ltcai 11.1.0 → 11.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (289) hide show
  1. package/README.md +47 -54
  2. package/docs/CHANGELOG.md +94 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/FEATURE_AUDIT_v11.2.0.md +393 -0
  6. package/docs/LAYOUT_REBUILD_SPEC.md +9 -1
  7. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  8. package/docs/ONBOARDING.md +1 -1
  9. package/docs/OPERATIONS.md +6 -2
  10. package/docs/PERMISSION_MODE.md +1 -1
  11. package/docs/TRUST_MODEL.md +1 -1
  12. package/docs/WHY_LATTICE.md +1 -1
  13. package/docs/architecture.md +6 -2
  14. package/docs/kg-schema.md +2 -2
  15. package/docs/v11.3.0_PLAN.md +202 -0
  16. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
  17. package/lattice_brain/__init__.py +1 -1
  18. package/lattice_brain/gates.py +125 -0
  19. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  20. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  21. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  22. package/lattice_brain/graph/_kg_common/text.py +479 -0
  23. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  24. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  25. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  26. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  27. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  28. package/lattice_brain/graph/fusion.py +35 -4
  29. package/lattice_brain/graph/projection/__init__.py +42 -0
  30. package/lattice_brain/graph/projection/curation.py +500 -0
  31. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +81 -485
  32. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  33. package/lattice_brain/graph/retrieval/context.py +197 -0
  34. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  35. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  36. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  37. package/lattice_brain/graph/retrieval/signals.py +95 -0
  38. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  39. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  40. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  41. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  42. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  43. package/lattice_brain/graph/schema.py +9 -0
  44. package/lattice_brain/graph/store.py +9 -0
  45. package/lattice_brain/graph/vector_index/selector.py +32 -2
  46. package/lattice_brain/ingestion/__init__.py +130 -0
  47. package/lattice_brain/ingestion/_contract.py +90 -0
  48. package/lattice_brain/ingestion/constants.py +127 -0
  49. package/lattice_brain/ingestion/folder_scan.py +57 -0
  50. package/lattice_brain/ingestion/folders.py +258 -0
  51. package/lattice_brain/ingestion/hashing.py +26 -0
  52. package/lattice_brain/ingestion/jobs_api.py +107 -0
  53. package/lattice_brain/ingestion/models.py +80 -0
  54. package/lattice_brain/ingestion/pipeline.py +486 -0
  55. package/lattice_brain/ingestion/quality.py +209 -0
  56. package/lattice_brain/ingestion/routing.py +295 -0
  57. package/lattice_brain/multimodal/__init__.py +164 -0
  58. package/lattice_brain/multimodal/audio.py +77 -0
  59. package/lattice_brain/multimodal/common.py +118 -0
  60. package/lattice_brain/{multimodal.py → multimodal/images.py} +22 -262
  61. package/lattice_brain/multimodal/ports.py +169 -0
  62. package/lattice_brain/multimodal/video.py +410 -0
  63. package/lattice_brain/portability/__init__.py +90 -0
  64. package/lattice_brain/portability/_contract.py +42 -0
  65. package/lattice_brain/portability/backups.py +338 -0
  66. package/lattice_brain/portability/bundles.py +136 -0
  67. package/lattice_brain/portability/constants.py +93 -0
  68. package/lattice_brain/portability/fsops.py +138 -0
  69. package/lattice_brain/portability/service.py +41 -0
  70. package/lattice_brain/portability/sharing.py +714 -0
  71. package/lattice_brain/runtime/__init__.py +1 -1
  72. package/lattice_brain/runtime/multi_agent.py +1 -1
  73. package/lattice_brain/sealed_box.py +244 -0
  74. package/lattice_brain/synthesis.py +24 -1
  75. package/latticeai/__init__.py +1 -1
  76. package/latticeai/api/brain_intelligence.py +4 -0
  77. package/latticeai/api/chat.py +11 -0
  78. package/latticeai/api/chat_helpers.py +16 -3
  79. package/latticeai/api/chat_hybrid.py +32 -1
  80. package/latticeai/api/chronicle.py +63 -0
  81. package/latticeai/api/features.py +70 -0
  82. package/latticeai/api/local_files.py +102 -0
  83. package/latticeai/api/portability.py +39 -4
  84. package/latticeai/api/review_queue.py +126 -0
  85. package/latticeai/api/search.py +16 -2
  86. package/latticeai/core/agent/__init__.py +93 -0
  87. package/latticeai/core/agent/_contract.py +79 -0
  88. package/latticeai/core/agent/context.py +57 -0
  89. package/latticeai/core/agent/deps.py +125 -0
  90. package/latticeai/core/agent/execution.py +622 -0
  91. package/latticeai/core/agent/planning.py +145 -0
  92. package/latticeai/core/agent/recovery.py +157 -0
  93. package/latticeai/core/agent/runtime.py +210 -0
  94. package/latticeai/core/agent/verification.py +231 -0
  95. package/latticeai/core/config.py +4 -1
  96. package/latticeai/core/context_builder.py +6 -3
  97. package/latticeai/core/embedding_providers/__init__.py +151 -0
  98. package/latticeai/core/embedding_providers/base.py +199 -0
  99. package/latticeai/core/embedding_providers/captions.py +162 -0
  100. package/latticeai/core/embedding_providers/profiles.py +126 -0
  101. package/latticeai/core/embedding_providers/text.py +350 -0
  102. package/latticeai/core/embedding_providers/vision.py +352 -0
  103. package/latticeai/core/file_generation/__init__.py +115 -0
  104. package/latticeai/core/file_generation/bundles.py +76 -0
  105. package/latticeai/core/file_generation/extraction.py +154 -0
  106. package/latticeai/core/file_generation/inference.py +235 -0
  107. package/latticeai/core/file_generation/orchestration.py +152 -0
  108. package/latticeai/core/file_generation/prompting.py +117 -0
  109. package/latticeai/core/file_generation/repair.py +114 -0
  110. package/latticeai/core/file_generation/sanitize.py +61 -0
  111. package/latticeai/core/file_generation/validation.py +201 -0
  112. package/latticeai/core/legacy_compatibility.py +1 -1
  113. package/latticeai/core/marketplace.py +1 -1
  114. package/latticeai/core/messages.py +152 -0
  115. package/latticeai/core/model_compat.py +73 -2
  116. package/latticeai/core/workspace_os_constants.py +1 -1
  117. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  118. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  119. package/latticeai/integrations/telegram_bot/config.py +86 -0
  120. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  121. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  122. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  123. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  124. package/latticeai/models/model_providers.py +12 -4
  125. package/latticeai/models/router/__init__.py +88 -0
  126. package/latticeai/models/router/_contract.py +66 -0
  127. package/latticeai/models/router/branding.py +56 -0
  128. package/latticeai/models/router/catalog.py +69 -0
  129. package/latticeai/models/router/documents.py +199 -0
  130. package/latticeai/models/router/errors.py +37 -0
  131. package/latticeai/models/router/generation.py +258 -0
  132. package/latticeai/models/router/loading.py +291 -0
  133. package/latticeai/models/router/local_models.py +85 -0
  134. package/latticeai/models/router/registry.py +147 -0
  135. package/latticeai/runtime/build_phases/__init__.py +82 -0
  136. package/latticeai/runtime/build_phases/features.py +407 -0
  137. package/latticeai/runtime/build_phases/foundation.py +555 -0
  138. package/latticeai/runtime/build_phases/web.py +492 -0
  139. package/latticeai/runtime/chat_wiring.py +4 -0
  140. package/latticeai/runtime/feature_toggle_wiring.py +163 -0
  141. package/latticeai/runtime/router_registration.py +11 -0
  142. package/latticeai/runtime/runtime_context.py +1 -0
  143. package/latticeai/services/app_context.py +8 -0
  144. package/latticeai/services/architecture_readiness.py +48 -19
  145. package/latticeai/services/automation_intelligence.py +22 -2
  146. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  147. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  148. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  149. package/latticeai/services/brain_intelligence/constants.py +47 -0
  150. package/latticeai/services/brain_intelligence/digest.py +258 -0
  151. package/latticeai/services/brain_intelligence/health.py +331 -0
  152. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  153. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  154. package/latticeai/services/brain_intelligence/service.py +48 -0
  155. package/latticeai/services/chronicle.py +557 -0
  156. package/latticeai/services/command_center.py +10 -4
  157. package/latticeai/services/feature_toggles.py +502 -0
  158. package/latticeai/services/folder_watch.py +122 -1
  159. package/latticeai/services/hybrid_chat.py +56 -5
  160. package/latticeai/services/interop_bridges.py +978 -0
  161. package/latticeai/services/memory_service/__init__.py +52 -0
  162. package/latticeai/services/memory_service/_contract.py +100 -0
  163. package/latticeai/services/memory_service/brief.py +431 -0
  164. package/latticeai/services/memory_service/constants.py +57 -0
  165. package/latticeai/services/memory_service/maintenance.py +138 -0
  166. package/latticeai/services/memory_service/manager.py +186 -0
  167. package/latticeai/services/memory_service/proof.py +136 -0
  168. package/latticeai/services/memory_service/recall.py +225 -0
  169. package/latticeai/services/memory_service/service.py +48 -0
  170. package/latticeai/services/memory_service/stores.py +110 -0
  171. package/latticeai/services/model_capability_registry.py +434 -261
  172. package/latticeai/services/model_catalog.py +95 -61
  173. package/latticeai/services/model_recommendation.py +18 -11
  174. package/latticeai/services/model_runtime/__init__.py +322 -0
  175. package/latticeai/services/model_runtime/cloud.py +87 -0
  176. package/latticeai/services/model_runtime/download.py +282 -0
  177. package/latticeai/services/model_runtime/engines.py +341 -0
  178. package/latticeai/services/model_runtime/loading.py +178 -0
  179. package/latticeai/services/model_runtime/service.py +129 -0
  180. package/latticeai/services/model_runtime/state.py +131 -0
  181. package/latticeai/services/model_runtime/status.py +255 -0
  182. package/latticeai/services/multimodal_ports.py +26 -1
  183. package/latticeai/services/obsidian_bridge.py +16 -25
  184. package/latticeai/services/product_readiness.py +15 -7
  185. package/latticeai/services/search_service.py +149 -2
  186. package/latticeai/services/tool_dispatch.py +4 -0
  187. package/latticeai/setup/auto_setup.py +27 -30
  188. package/latticeai/setup/wizard/__init__.py +126 -0
  189. package/latticeai/setup/wizard/catalog.py +172 -0
  190. package/latticeai/setup/wizard/detect.py +323 -0
  191. package/latticeai/setup/wizard/install.py +348 -0
  192. package/latticeai/setup/wizard/paths.py +168 -0
  193. package/latticeai/setup/wizard/plans.py +74 -0
  194. package/latticeai/setup/wizard/recommend.py +320 -0
  195. package/package.json +6 -2
  196. package/scripts/bump_version.py +14 -0
  197. package/scripts/capture_release_evidence.mjs +33 -21
  198. package/scripts/check_current_release_docs.mjs +1 -1
  199. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  200. package/scripts/check_max_file_lines.mjs +102 -0
  201. package/scripts/check_release_evidence_bound.mjs +30 -15
  202. package/scripts/check_screenshot_pixel_delta.py +34 -4
  203. package/scripts/check_server_i18n.mjs +2 -0
  204. package/scripts/generate_rust_parity_fixtures.py +562 -0
  205. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  206. package/scripts/release_screen_claims.json +44 -2
  207. package/scripts/verify_hf_model_registry.py +253 -218
  208. package/src-tauri/Cargo.lock +361 -3
  209. package/src-tauri/Cargo.toml +6 -1
  210. package/src-tauri/src/backend.rs +349 -0
  211. package/src-tauri/src/folder.rs +33 -0
  212. package/src-tauri/src/main.rs +97 -399
  213. package/src-tauri/tauri.conf.json +1 -1
  214. package/static/app/asset-manifest.json +41 -37
  215. package/static/app/assets/Act-yYpYnn0v.js +1 -0
  216. package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
  217. package/static/app/assets/{Brain-CzCsI1mi.js → Brain-C1HBN0Wf.js} +2 -2
  218. package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
  219. package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
  220. package/static/app/assets/Capture-CFIRsFNE.js +1 -0
  221. package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
  222. package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
  223. package/static/app/assets/Library-DwO3yZST.js +1 -0
  224. package/static/app/assets/{LivingBrain-BXMWIK_2.js → LivingBrain-Jn1GK0-S.js} +1 -1
  225. package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
  226. package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
  227. package/static/app/assets/System-DW8F-2xL.js +1 -0
  228. package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
  229. package/static/app/assets/{bot-4BvN07ux.js → bot-IM_E_Y12.js} +1 -1
  230. package/static/app/assets/brain-Ci1CkWjM.js +1 -0
  231. package/static/app/assets/{button-CDjtnAoU.js → button-COwyqfHM.js} +1 -1
  232. package/static/app/assets/circle-check-DfInj-qD.js +1 -0
  233. package/static/app/assets/{circle-pause-D_RMn7tp.js → circle-pause-DEM4A1Y5.js} +1 -1
  234. package/static/app/assets/{circle-play-B5OpB8ae.js → circle-play-C9djDuLd.js} +1 -1
  235. package/static/app/assets/{cpu-BIlWInHf.js → cpu-DFdo1gw-.js} +1 -1
  236. package/static/app/assets/{download-BtjXfL3z.js → download-SnJL6oqk.js} +1 -1
  237. package/static/app/assets/{folder-open-DefMpxI2.js → folder-open-CqZeDkjE.js} +1 -1
  238. package/static/app/assets/{hard-drive-BQ8NZVkw.js → hard-drive-j1jJXYYf.js} +1 -1
  239. package/static/app/assets/{index-vtEfYvQY.css → index-BLPb5lmE.css} +1 -1
  240. package/static/app/assets/index-_u5iUHDr.js +10 -0
  241. package/static/app/assets/input-B0lPdRQZ.js +1 -0
  242. package/static/app/assets/link-2-CoFbooHS.js +1 -0
  243. package/static/app/assets/{permissionCopy-BqZ5tsgL.js → permissionCopy-BsyLxtao.js} +1 -1
  244. package/static/app/assets/primitives-DEbN-d6p.js +1 -0
  245. package/static/app/assets/search-BybIWPNd.js +1 -0
  246. package/static/app/assets/{share-2-D5zg_0fY.js → share-2-CVtZ_ewX.js} +1 -1
  247. package/static/app/assets/{shield-alert-B5pZzkUb.js → shield-alert-CBi2GNWM.js} +1 -1
  248. package/static/app/assets/{textarea-nEVIweKY.js → textarea-DNMpB5ih.js} +1 -1
  249. package/static/app/assets/{useFocusTrap-Cm99AHlz.js → useFocusTrap-C83t3GXF.js} +1 -1
  250. package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
  251. package/static/app/assets/{useQuery-Dm__N6bL.js → useQuery-Dcp1OChy.js} +1 -1
  252. package/static/app/assets/utils-BlZr7Pd4.js +4 -0
  253. package/static/app/assets/workspace-jJY4RuAV.js +1 -0
  254. package/static/app/index.html +4 -4
  255. package/static/sw.js +1 -1
  256. package/lattice_brain/graph/_kg_common.py +0 -1331
  257. package/lattice_brain/graph/discovery_index.py +0 -1141
  258. package/lattice_brain/graph/retrieval.py +0 -1120
  259. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  260. package/lattice_brain/ingestion.py +0 -1377
  261. package/lattice_brain/portability.py +0 -1210
  262. package/latticeai/core/agent.py +0 -1412
  263. package/latticeai/core/embedding_providers.py +0 -1196
  264. package/latticeai/core/file_generation.py +0 -1047
  265. package/latticeai/integrations/telegram_bot.py +0 -1390
  266. package/latticeai/models/router.py +0 -1007
  267. package/latticeai/runtime/build_phases.py +0 -1422
  268. package/latticeai/services/brain_intelligence.py +0 -967
  269. package/latticeai/services/memory_service.py +0 -1177
  270. package/latticeai/services/model_runtime.py +0 -1281
  271. package/latticeai/setup/wizard.py +0 -1277
  272. package/static/app/assets/Act-D0HWqtn0.js +0 -1
  273. package/static/app/assets/AdminConsole-D-QDW-A4.js +0 -1
  274. package/static/app/assets/BrainHome-Btns-_TA.js +0 -2
  275. package/static/app/assets/BrainSignals-2dHQNkns.js +0 -1
  276. package/static/app/assets/Capture-CT8v1StE.js +0 -1
  277. package/static/app/assets/CommandPalette-DoLXC2KH.js +0 -1
  278. package/static/app/assets/Library-DDoxFE5c.js +0 -1
  279. package/static/app/assets/ProductFlow-DOYf7JIs.js +0 -1
  280. package/static/app/assets/ReviewCard-COQsqidK.js +0 -3
  281. package/static/app/assets/System-BRllvYXd.js +0 -1
  282. package/static/app/assets/arrow-left-DnyMzss-.js +0 -1
  283. package/static/app/assets/brain-uMb_5hnO.js +0 -1
  284. package/static/app/assets/index-0AvoEBzJ.js +0 -10
  285. package/static/app/assets/input-B_5ZJ9oy.js +0 -1
  286. package/static/app/assets/primitives-CVwew78r.js +0 -1
  287. package/static/app/assets/search-DkhnOKZt.js +0 -1
  288. package/static/app/assets/utils-DcDMoZIe.js +0 -4
  289. package/static/app/assets/workspace-LtRRSKTf.js +0 -1
@@ -1,1412 +0,0 @@
1
- """Single-agent runtime — the Discover→Plan→Implement→Verify state machine.
2
-
3
- This module is the deep single-agent loop: a small interface (``AgentDeps`` ports +
4
- ``SingleAgentRuntime.run_to_completion``) over the whole role-phased state machine
5
- (planner → executor → critic → rollback → memory). It carries no FastAPI,
6
- no globals, and no I/O of its own — every collaborator is injected through
7
- ``AgentDeps``.
8
-
9
- Two adapters justify the seam:
10
-
11
- * production wires ``AgentDeps`` from ``latticeai.server_app``'s ``LLMRouter``, governance
12
- map, audit log, and prompts;
13
- * tests pass fake ports (an LLM that returns canned JSON, a recording tool
14
- executor) and drive a full PLAN→EXECUTE→VERIFY→DONE cycle without a server.
15
-
16
- HTTP concerns — request parsing, chat-history persistence, response shaping,
17
- scheduling the background memory update — stay in the app layer. This module
18
- only owns the state machine.
19
- """
20
-
21
- from __future__ import annotations
22
-
23
- import json
24
- import logging
25
- from dataclasses import dataclass
26
- from pathlib import Path
27
- from typing import (
28
- Any,
29
- Awaitable,
30
- Callable,
31
- Dict,
32
- FrozenSet,
33
- List,
34
- Mapping,
35
- Optional,
36
- Tuple,
37
- )
38
-
39
- from lattice_brain.runtime.contracts import (
40
- runtime_boundary_contract,
41
- single_agent_contract,
42
- )
43
- from lattice_brain.runtime.hooks import dispatch_tool
44
- from latticeai.core.agent_helpers import (
45
- PhaseBudgets,
46
- TranscriptBudget,
47
- _truncate_strings,
48
- artifact_checklist,
49
- compact_transcript,
50
- extract_action,
51
- extract_action_details,
52
- files_written,
53
- filter_learnings,
54
- format_artifact_checklist,
55
- format_requirement_coverage,
56
- normalize_plan,
57
- requirement_coverage,
58
- )
59
- from latticeai.core.agent_permission import (
60
- block_reason_for_tool,
61
- non_auto_plan_steps,
62
- resolve_deps_mode,
63
- )
64
- from latticeai.core.agent_profiles import AgentProfile, profile_for_model
65
- from latticeai.core.agent_prompts import executor_prompt_for
66
-
67
- # The state vocabulary and the pure helpers live in sibling modules so this one
68
- # holds only the loop. They are re-exported (see ``__all__``) because callers —
69
- # the HTTP layer, run_store, the eval harness, and the tests — have always
70
- # imported them from here, and that contract does not change.
71
- from latticeai.core.agent_state import AGENT_TERMINAL_STATES, AgentState
72
- from latticeai.core.agent_trace import LoopTrace
73
- from latticeai.core.file_generation import (
74
- generate_file_content,
75
- infer_file_target,
76
- sanitize_write_content,
77
- )
78
- from latticeai.core.permission_mode import (
79
- PermissionMode,
80
- is_circuit_breaker,
81
- plan_requires_approval,
82
- should_stage_proposal,
83
- )
84
- from latticeai.core.tool_governor import classify_tool_call
85
- from latticeai.core.tool_registry import SCOPED_KNOWLEDGE_TOOLS
86
- from latticeai.tools import ToolError, document_output_target
87
-
88
- __all__ = [
89
- # this module
90
- "AgentDeps",
91
- "AgentRunContext",
92
- "SingleAgentRuntime",
93
- # re-exported from agent_state
94
- "AGENT_TERMINAL_STATES",
95
- "AgentState",
96
- # re-exported from agent_helpers
97
- "PhaseBudgets",
98
- "TranscriptBudget",
99
- "artifact_checklist",
100
- "compact_transcript",
101
- "extract_action",
102
- "extract_action_details",
103
- "files_written",
104
- "filter_learnings",
105
- "format_artifact_checklist",
106
- "format_requirement_coverage",
107
- "normalize_plan",
108
- "requirement_coverage",
109
- ]
110
-
111
-
112
- class AgentRunContext:
113
- """Mutable state carrier passed through all agent phases."""
114
- __slots__ = ("state", "plan", "transcript", "retry_count",
115
- "state_history", "corrections", "final_message", "rollback_log",
116
- "executing_model", "reviewing_model", "approved_by_human", "trace",
117
- "on_step", "project_context", "permission_mode")
118
-
119
- def __init__(self) -> None:
120
- self.state: AgentState = AgentState.IDLE
121
- self.trace: LoopTrace = LoopTrace()
122
- self.plan: dict = {}
123
- self.transcript: list = []
124
- self.retry_count: int = 0
125
- self.state_history: list = []
126
- self.corrections: list = []
127
- self.final_message: str = ""
128
- self.rollback_log: list = []
129
- self.executing_model: Optional[str] = None
130
- self.reviewing_model: Optional[str] = None
131
- self.approved_by_human: bool = False
132
- # Per-run step observer (review Wave 1.1): the HTTP layer attaches a
133
- # callback here so live SSE clients see progress while EXECUTING.
134
- # Never serialized; a broken observer never breaks the loop.
135
- self.on_step: Optional[Callable[[Dict[str, Any]], None]] = None
136
- # Multi-turn project loop (v9.9.6): a prompt block describing where the
137
- # project stands — files already produced, open TODOs, the last honest
138
- # verification. Empty for a standalone run, which behaves exactly as
139
- # before. Set by the HTTP layer, read by plan/execute/verify.
140
- self.project_context: str = ""
141
- # Autonomy dial resolved once per run (v9.9.8). The HTTP layer stamps
142
- # the user/workspace-scoped mode here so the plan gate and every
143
- # per-tool gate in the same run agree; ``None`` falls back to the
144
- # process-wide resolver on ``deps``.
145
- self.permission_mode: Optional[str] = None
146
-
147
-
148
- @dataclass
149
- class AgentDeps:
150
- """The ports a :class:`SingleAgentRuntime` needs from the outside world.
151
-
152
- Everything the state machine touches is here, so the loop can be exercised
153
- against fakes. See module docstring for the two-adapter rationale.
154
- """
155
-
156
- # ── LLM port ─────────────────────────────────────────────────────
157
- # generate_as(model_id, message, context, max_tokens, temperature) -> str
158
- generate_as: Callable[..., Awaitable[Any]]
159
- # generate(message, context, max_tokens, temperature) -> str
160
- generate: Callable[..., Awaitable[Any]]
161
-
162
- # ── tool port ────────────────────────────────────────────────────
163
- execute_tool: Callable[[str, dict], dict]
164
- policy_for: Callable[[str, dict], Mapping[str, Any]] # name, args -> policy
165
- risk_level: Callable[[Any], str] # policy -> "low"|"medium"|"high"
166
- check_role: Callable[[str, str], None] # tool_name, user -> raises if not allowed
167
- tool_governance: Mapping[str, Mapping[str, Any]] # name -> policy (auto_approve set)
168
- file_create_actions: FrozenSet[str]
169
-
170
- # ── context / memory / audit ports ───────────────────────────────
171
- recent_chat_context: Callable[..., str] # (conversation_id=...) -> str
172
- clear_history: Callable[[int], dict]
173
- knowledge_save: Callable[..., Any]
174
- audit: Callable[..., None] # (event, **kw) -> None
175
-
176
- # ── prompts + config ─────────────────────────────────────────────
177
- planner_prompt: str
178
- executor_prompt: str
179
- critic_prompt: str
180
- memory_updater_prompt: str
181
- agent_root: Path
182
-
183
- # ── rollback port (optional) ─────────────────────────────────────
184
- # Production injects this from the tool dispatch service so this pure
185
- # state machine does not shell out directly. Tests can pass a recorder.
186
- rollback_file: Optional[Callable[[str], Dict[str, Any]]] = None
187
-
188
- # ── snapshot rollback ports (optional, review L7) ────────────────
189
- # git-only rollback left non-git workspaces and newly created files
190
- # unrecoverable. ``snapshot_file(path)`` captures pre-write state
191
- # ({"existed", "content", "too_large"}) before a file-create action;
192
- # ``restore_snapshot(path, content)`` restores it (content=None deletes
193
- # a file the run created). Both are production-wired with workspace
194
- # path safety; tests pass recorders.
195
- snapshot_file: Optional[Callable[[str], Dict[str, Any]]] = None
196
- restore_snapshot: Optional[Callable[[str, Optional[str]], Dict[str, Any]]] = None
197
-
198
- # ── lifecycle hooks port (optional) ──────────────────────────────
199
- # When present, every tool execution fires the shared pre_tool/post_tool
200
- # lifecycle, so the agent tool path no longer bypasses hooks.
201
- hooks: Any = None
202
-
203
- # ── brain memory port (optional) ─────────────────────────────────
204
- # When present, completed-run learnings become typed Experience records
205
- # through the unified ingestion pipeline (with provenance), replacing
206
- # the vault markdown dump.
207
- brain_memory: Any = None
208
-
209
- # ── change governor port (optional) ──────────────────────────────
210
- # When present, file writes are classified centrally: additive creates
211
- # run with minimal friction, while mutations/deletions of existing
212
- # content are staged as review proposals instead of applied. The port is
213
- # ``review(name, args, policy=..., user_email=..., workspace_id=...)``
214
- # returning None (fall through to the classic gates) or a verdict dict.
215
- change_governor: Any = None
216
-
217
- # ── phase budgets (optional) ─────────────────────────────────────
218
- # Per-phase token caps (plan/execute/verify/memory). None reads the
219
- # environment once at first use; tests inject a fixed PhaseBudgets.
220
- phase_budgets: Optional[PhaseBudgets] = None
221
-
222
- # ── transcript shaping (optional) ────────────────────────────────
223
- # Executor/critic prompt window caps. None reads the environment once;
224
- # tests inject a fixed TranscriptBudget.
225
- transcript_budget: Optional["TranscriptBudget"] = None
226
-
227
- # ── step observer port (optional) ────────────────────────────────
228
- # Default per-runtime observer for live step events; a per-run observer
229
- # can also be attached on AgentRunContext.on_step. Both are advisory.
230
- on_step: Optional[Callable[[Dict[str, Any]], None]] = None
231
-
232
- # ── agent profile (optional, v9.9.7) ─────────────────────────────
233
- # How hard the loop works to keep a weak model on contract. None selects
234
- # per-run from the executing model id (``profile_for_model``); tests
235
- # inject a fixed profile.
236
- agent_profile: Optional["AgentProfile"] = None
237
-
238
- # ── permission mode port (optional, v9.9.8) ──────────────────────
239
- # The autonomy dial. Either a static mode, or a resolver callable that
240
- # accepts ``user_email``/``workspace_id`` scope kwargs (preferred) or no
241
- # arguments at all — see ``call_mode_source``. ``None`` means strict,
242
- # which is exactly the pre-9.9.8 behaviour.
243
- permission_mode: Any = None
244
-
245
-
246
- class SingleAgentRuntime:
247
- """Drives the agent state machine over injected :class:`AgentDeps`."""
248
-
249
- def __init__(self, deps: AgentDeps) -> None:
250
- self.deps = deps
251
- self._env_phase_budgets: Optional[PhaseBudgets] = None
252
- self._env_transcript_budget: Optional[TranscriptBudget] = None
253
-
254
- @property
255
- def phase_budgets(self) -> PhaseBudgets:
256
- # getattr twice: partially-constructed runtimes/deps (tests build them
257
- # via __new__ or minimal fakes) still get working default budgets.
258
- injected = getattr(self.deps, "phase_budgets", None)
259
- if injected is not None:
260
- return injected
261
- cached = getattr(self, "_env_phase_budgets", None)
262
- if cached is None:
263
- cached = PhaseBudgets.from_env()
264
- self._env_phase_budgets = cached
265
- return cached
266
-
267
- @property
268
- def transcript_budget(self) -> TranscriptBudget:
269
- injected = getattr(self.deps, "transcript_budget", None)
270
- if injected is not None:
271
- return injected
272
- cached = getattr(self, "_env_transcript_budget", None)
273
- if cached is None:
274
- cached = TranscriptBudget.from_env()
275
- self._env_transcript_budget = cached
276
- return cached
277
-
278
- # ── permission mode (v9.9.8) ─────────────────────────────────────
279
- def resolve_permission_mode(
280
- self,
281
- ctx: Optional[AgentRunContext] = None,
282
- *,
283
- user_email: Optional[str] = None,
284
- workspace_id: Optional[str] = None,
285
- ) -> PermissionMode:
286
- """Autonomy dial for this run.
287
-
288
- A mode stamped on ``ctx`` wins, so the plan a user approved and every
289
- tool step in the same run are judged by one dial even if the stored
290
- preference changes mid-run. Otherwise the resolver on ``deps`` is
291
- consulted with the caller's scope — resolving unscoped would collapse
292
- every caller onto the process-wide default.
293
- """
294
- return resolve_deps_mode(
295
- self.deps, ctx, user_email=user_email, workspace_id=workspace_id,
296
- )
297
-
298
- def _governed_path_exists(self, name: str, path: str) -> bool:
299
- """Does this tool call's *real* target already exist?
300
-
301
- The document creators sanitize ``filename`` into their own output
302
- directory, so the raw argument is resolved through
303
- :func:`document_output_target` first — checking it verbatim would
304
- inspect a path nothing ever writes and the fail-closed overwrite guard
305
- would never fire. Workspace-relative paths resolve under
306
- ``deps.agent_root``; absolute paths (home-sandbox writes) are honored
307
- as-is. Never raises: governance must not be able to crash the loop, and
308
- an unresolvable path degrades to "new file", which the remaining gates
309
- still cover.
310
- """
311
- try:
312
- candidate = Path(document_output_target(name, path) or path)
313
- if not candidate.is_absolute():
314
- candidate = Path(self.deps.agent_root) / candidate
315
- return candidate.exists()
316
- except Exception: # noqa: BLE001 — classification is best-effort
317
- return False
318
-
319
- def _governed_tools(self) -> FrozenSet[str]:
320
- governor = getattr(self.deps, "change_governor", None)
321
- if governor is None:
322
- return frozenset()
323
- return frozenset(getattr(governor, "governed_tools", frozenset()))
324
-
325
- def profile_for(self, model_id: Optional[str]) -> AgentProfile:
326
- """Loop profile for the model actually executing this run (v9.9.7).
327
-
328
- An injected ``AgentDeps.agent_profile`` wins (tests, explicit config);
329
- otherwise the profile is derived from the model id, so a small local
330
- model gets the compact loop without any extra configuration.
331
- """
332
- injected = getattr(self.deps, "agent_profile", None)
333
- if injected is not None:
334
- return injected
335
- return profile_for_model(model_id)
336
-
337
- def _emit_step(self, ctx: AgentRunContext, phase: str, event: str, **details: Any) -> None:
338
- """Fire the per-run / deps step observers (review Wave 1.1).
339
-
340
- Observers power the live step timeline in the UI. They are pure
341
- telemetry: any observer failure is logged and swallowed — the loop
342
- itself must never notice.
343
- """
344
- payload: Dict[str, Any] = {"phase": phase, "event": event}
345
- for key, value in details.items():
346
- if value is not None:
347
- payload[key] = value
348
- for observer in (getattr(ctx, "on_step", None), getattr(self.deps, "on_step", None)):
349
- if observer is None:
350
- continue
351
- try:
352
- observer(dict(payload))
353
- except Exception as exc: # noqa: BLE001 — observers are advisory
354
- logging.warning("agent step observer failed: %s", exc)
355
-
356
- @staticmethod
357
- def _project_block(ctx: AgentRunContext) -> str:
358
- """Project-session context for prompts, or "" for a standalone run.
359
-
360
- Multi-turn project loop (v9.9.6): a later run must see the files the
361
- project already produced and what is still open, instead of planning
362
- from a blank workspace every time.
363
- """
364
- summary = str(getattr(ctx, "project_context", "") or "").strip()
365
- return f"\n\n[PROJECT SESSION]\n{summary}" if summary else ""
366
-
367
- def boundary(self) -> Dict[str, Any]:
368
- return runtime_boundary_contract(
369
- name="SingleAgentRuntime",
370
- runtime="single_agent",
371
- entrypoint="latticeai.core.agent.SingleAgentRuntime",
372
- surface="/agent",
373
- owns="single-agent PLAN / EXECUTE / VERIFY state machine over injected ports",
374
- compatibility_aliases=[],
375
- )
376
-
377
- def config(self) -> Dict[str, Any]:
378
- return {
379
- "boundary": self.boundary(),
380
- "states": [state.value for state in AgentState],
381
- "terminal_states": sorted(state.value for state in AGENT_TERMINAL_STATES),
382
- "execution_mode": "injected_ports",
383
- }
384
-
385
- def contract(self, ctx: AgentRunContext, req: Any, *, run_id: Optional[str] = None) -> Dict[str, Any]:
386
- """Expose the shared agent-run contract for the single-agent loop."""
387
- return single_agent_contract(ctx=ctx, goal=getattr(req, "message", ""), run_id=run_id)
388
-
389
- # ── PLAN ─────────────────────────────────────────────────────────
390
- async def plan(
391
- self, ctx: AgentRunContext, req: Any, lang_hint: str, current_user: str,
392
- model_id: Optional[str] = None,
393
- ) -> None:
394
- """PLAN: Planner role produces a structured plan JSON."""
395
- d = self.deps
396
- project_block = self._project_block(ctx)
397
- context = (
398
- f"{d.planner_prompt}\n\n"
399
- f"[LANGUAGE HINT: {lang_hint}]\n"
400
- f"Workspace root: {d.agent_root}{project_block}\n\n"
401
- f"User request: {req.message}"
402
- )
403
- raw = await d.generate_as(
404
- model_id,
405
- message="Produce a JSON execution plan for this request.",
406
- context=context, max_tokens=self.phase_budgets.plan_tokens, temperature=0.1,
407
- )
408
- ctx.trace.llm_call("plan", model=model_id)
409
- try:
410
- plan, plan_repairs = extract_action_details(str(raw))
411
- ctx.trace.repair("plan", repairs=plan_repairs)
412
- except ValueError as exc:
413
- ctx.trace.parse_error("plan", error=str(exc), recovered=True)
414
- plan = {
415
- "action": "plan", "state": "PLAN",
416
- "goal": req.message, "steps": [],
417
- "requires_approval": False, "rollback_strategy": "none", "estimated_steps": 1,
418
- }
419
- plan, plan_fixes = normalize_plan(plan, req.message)
420
- if plan_fixes:
421
- ctx.trace.repair("plan", repairs=plan_fixes)
422
- ctx.plan = plan
423
- ctx.transcript.append({
424
- "state": AgentState.PLANNING.value,
425
- "goal": plan.get("goal", req.message),
426
- "steps": plan.get("steps", []),
427
- "requires_approval": plan.get("requires_approval", False),
428
- "rollback_strategy": plan.get("rollback_strategy", "none"),
429
- "estimated_steps": plan.get("estimated_steps", 1),
430
- **({"plan_fixes": plan_fixes} if plan_fixes else {}),
431
- })
432
- self._emit_step(
433
- ctx, "plan", "planned",
434
- goal=str(plan.get("goal") or "")[:200],
435
- steps=len(plan.get("steps") or []),
436
- requires_approval=bool(plan.get("requires_approval", False)),
437
- )
438
- ctx.state = AgentState.WAITING_APPROVAL
439
-
440
- # ── APPROVAL ─────────────────────────────────────────────────────
441
- def approval_requirements(self, ctx: AgentRunContext) -> Dict[str, Any]:
442
- """Read-only preview of the approval gate for a planned run.
443
-
444
- Shares the exact predicate :meth:`approve` enforces, so the HTTP
445
- layer can pause a run as ``awaiting_approval`` (with a plan summary
446
- for the user) instead of letting it fail closed — without ever
447
- weakening the gate itself.
448
- """
449
- d = self.deps
450
- mode = self.resolve_permission_mode(ctx)
451
- # Governor-managed tools never hard-block the plan: each call is
452
- # classified at execution time — additive creates run, mutations and
453
- # deletions of existing content become review proposals.
454
- governed_tools = self._governed_tools()
455
- steps = ctx.plan.get("steps", [])
456
- non_auto = non_auto_plan_steps(
457
- mode, steps, d.tool_governance or {}, governed_tools=governed_tools,
458
- )
459
- requires = plan_requires_approval(
460
- mode,
461
- non_auto_steps=non_auto,
462
- plan_flag=bool(ctx.plan.get("requires_approval", False)),
463
- )
464
- lines = [
465
- f"{index}. {step.get('description') or step.get('action') or '?'}"
466
- for index, step in enumerate(steps, start=1)
467
- ]
468
- summary = str(ctx.plan.get("goal") or "").strip()
469
- if lines:
470
- summary = (summary + "\n" if summary else "") + "\n".join(lines)
471
- return {
472
- "requires_approval": requires,
473
- "non_auto_steps": non_auto,
474
- "permission_mode": mode.value,
475
- "plan_summary": summary,
476
- }
477
-
478
- def approve(self, ctx: AgentRunContext, current_user: str, *, approved_by_human: bool = False) -> None:
479
- """APPROVAL: Check governance, log decision, auto-approve (future: UI prompt)."""
480
- d = self.deps
481
- requirements = self.approval_requirements(ctx)
482
- non_auto = requirements["non_auto_steps"]
483
- requires = requirements["requires_approval"]
484
-
485
- ctx.transcript.append({
486
- "state": AgentState.WAITING_APPROVAL.value,
487
- "requires_approval": requires,
488
- "non_auto_approve_steps": non_auto,
489
- "decision": "human_approved" if requires and approved_by_human else ("blocked_pending_approval" if requires else "auto_approved"),
490
- })
491
- decision = "human_approved" if requires and approved_by_human else ("blocked_pending_approval" if requires else "auto_approved")
492
- ctx.trace.decision("approve", decision=decision, non_auto_steps=len(non_auto))
493
- self._emit_step(ctx, "approval", "decision", decision=decision)
494
- d.audit(
495
- "agent_approval", user_email=current_user,
496
- requires_approval=requires,
497
- non_auto_steps=non_auto,
498
- decision=decision,
499
- )
500
- if requires and not approved_by_human:
501
- ctx.final_message = (
502
- "이 작업에는 명시 승인이 필요한 도구가 포함되어 있어 자동 실행을 중단했습니다. "
503
- "human_in_loop 승인 흐름으로 다시 실행해 주세요."
504
- )
505
- ctx.state = AgentState.FAILED
506
- return
507
- ctx.approved_by_human = bool(approved_by_human)
508
- ctx.state = AgentState.EXECUTING
509
-
510
- # ── EXECUTE ──────────────────────────────────────────────────────
511
- async def execute(
512
- self, ctx: AgentRunContext, req: Any, lang_hint: str,
513
- current_user: str, max_steps: int, model_id: Optional[str] = None,
514
- ) -> None:
515
- """EXECUTE: Executor role calls tools one at a time until final or budget exhausted."""
516
- d = self.deps
517
- profile = self.profile_for(model_id)
518
- exec_count = sum(1 for s in ctx.transcript if s.get("state") == AgentState.EXECUTING.value)
519
- budget = max(1, max_steps - exec_count)
520
- parse_failures = 0
521
-
522
- for _ in range(budget):
523
- request_workspace = getattr(req, "workspace_id", None)
524
- context = self._executor_context(
525
- ctx, req, lang_hint, current_user, request_workspace, profile=profile
526
- )
527
- raw = await d.generate_as(
528
- model_id,
529
- message="Execute the next step.",
530
- context=context, max_tokens=self.phase_budgets.execute_tokens,
531
- temperature=req.temperature,
532
- )
533
- ctx.trace.llm_call("execute", model=model_id)
534
- try:
535
- action, exec_repairs = extract_action_details(str(raw))
536
- ctx.trace.repair("execute", repairs=exec_repairs)
537
- except ValueError as exc:
538
- parse_failures += 1
539
- if self._note_parse_failure(ctx, raw, exc, parse_failures, profile):
540
- # Direct-path fallback (v9.9.7): a small model that cannot
541
- # hold the tool-call protocol can still write a file. Run
542
- # the plan's own file steps without asking for any JSON.
543
- if profile.direct_path_fallback and await self._direct_file_path(
544
- ctx, req, current_user, model_id
545
- ):
546
- ctx.state = AgentState.VERIFYING
547
- return
548
- break
549
- continue
550
-
551
- name = str(action.get("action") or "")
552
- thoughts = str(action.get("thoughts") or "")[:600]
553
- args = action.get("args") or {}
554
-
555
- if name in SCOPED_KNOWLEDGE_TOOLS:
556
- # Scope is server-owned, never model-owned. Overwrite any
557
- # claimed values before policy evaluation, audit, and dispatch.
558
- args = dict(args)
559
- args["workspace_id"] = request_workspace or "personal"
560
- args["user_email"] = current_user or "local"
561
-
562
- if name == "final":
563
- ctx.final_message = action.get("message", "작업을 완료했습니다.")
564
- ctx.transcript.append({
565
- "state": AgentState.EXECUTING.value, "action": "final", "thoughts": thoughts,
566
- })
567
- ctx.trace.decision("execute", decision="final")
568
- self._emit_step(ctx, "execute", "final")
569
- ctx.state = AgentState.VERIFYING
570
- return
571
-
572
- # Loop guard
573
- if self._is_repeated_create(ctx, name, args):
574
- ctx.transcript.append({
575
- "state": AgentState.EXECUTING.value, "action": name,
576
- "error": "LOOP_DETECTED: identical action+args repeated — halted.",
577
- })
578
- ctx.trace.decision("execute", decision="loop_detected", tool=name)
579
- self._emit_step(ctx, "execute", "blocked", action=name, reason="loop_detected")
580
- break
581
-
582
- if name == "clear_history":
583
- result = d.clear_history(args.get("keep_last", 0))
584
- ctx.transcript.append({
585
- "state": AgentState.EXECUTING.value, "action": name,
586
- "thoughts": thoughts, "args": args, "result": result,
587
- })
588
- self._emit_step(ctx, "execute", "tool", action=name, ok=True)
589
- continue
590
-
591
- policy = d.policy_for(name, args)
592
- risk = d.risk_level(policy)
593
-
594
- proposed, governor_allows_additive = self._governor_review(
595
- ctx, name, thoughts, args, policy, risk, current_user, request_workspace,
596
- conversation_id=getattr(req, "conversation_id", None),
597
- )
598
- if proposed:
599
- continue
600
-
601
- if self._blocked_by_gates(
602
- ctx, req, name, thoughts, args, policy, risk,
603
- current_user, governor_allows_additive,
604
- ):
605
- continue
606
-
607
- self._dispatch_step(ctx, name, thoughts, args, policy, risk, current_user)
608
-
609
- ctx.state = AgentState.VERIFYING
610
-
611
- def _executor_context(
612
- self, ctx: AgentRunContext, req: Any, lang_hint: str,
613
- current_user: str, request_workspace: Optional[str],
614
- profile: Optional[AgentProfile] = None,
615
- ) -> str:
616
- """Assemble one executor turn's prompt (plan, corrections, recent chat)."""
617
- d = self.deps
618
- # Only the latest corrections steer the next attempt — stale hints
619
- # from earlier retries dilute weak models (review Wave 0.3).
620
- active_corrections = ctx.corrections[-3:]
621
- corrections_hint = (
622
- "\n\nCritic corrections from previous attempt:\n"
623
- + "\n".join(f"- {c}" for c in active_corrections)
624
- ) if active_corrections else ""
625
-
626
- recent_kwargs = {
627
- "conversation_id": req.conversation_id,
628
- "user_email": current_user or None,
629
- }
630
- if request_workspace is not None:
631
- recent_kwargs["workspace_id"] = request_workspace
632
- recent_conversation = d.recent_chat_context(**recent_kwargs) or "(none)"
633
- budget = self.transcript_budget
634
- # A small model drowns in a long transcript far sooner than a large
635
- # one, so the profile may narrow the window (v9.9.7).
636
- window = min(budget.window, profile.transcript_window) if profile else budget.window
637
- bounded_transcript = compact_transcript(
638
- ctx.transcript,
639
- window=window,
640
- result_chars=budget.result_chars,
641
- )
642
- # Mid-run workspace awareness (review L5): later steps must see what
643
- # this run already produced instead of a stale workspace picture.
644
- written = files_written(ctx.transcript, d.file_create_actions)
645
- written_hint = (
646
- "\n\nFiles written by this run so far (they exist in the workspace now):\n"
647
- + "\n".join(f"- {path}" for path in written)
648
- ) if written else ""
649
- return (
650
- # v11.1.0: the executor prompt carries profile-aware file-writing
651
- # hints, because "wrote nothing at all" was the weak-model failure
652
- # mode the loop could not repair after the fact.
653
- f"{executor_prompt_for(d.executor_prompt, profile=profile)}\n\n"
654
- f"[LANGUAGE HINT: {lang_hint}]\n"
655
- f"Workspace root: {d.agent_root}{self._project_block(ctx)}\n\n"
656
- f"PLAN:\n{json.dumps(ctx.plan, ensure_ascii=False)}{written_hint}\n\n"
657
- f"Recent conversation:\n{recent_conversation}\n\n"
658
- f"User request: {req.message}{corrections_hint}\n\n"
659
- f"Execution transcript:\n{json.dumps(bounded_transcript, ensure_ascii=False, indent=2)}"
660
- )
661
-
662
- def _note_parse_failure(
663
- self, ctx: AgentRunContext, raw: Any, exc: ValueError, parse_failures: int,
664
- profile: Optional[AgentProfile] = None,
665
- ) -> bool:
666
- """Record one executor parse slip; True when the run should stop retrying."""
667
- profile = profile or self.profile_for(None)
668
- ctx.transcript.append({
669
- "state": AgentState.EXECUTING.value, "action": "parse_error",
670
- "raw": str(raw)[:400], "error": str(exc),
671
- })
672
- if parse_failures >= profile.parse_failure_budget:
673
- ctx.trace.parse_error("execute", error=str(exc), recovered=False)
674
- self._emit_step(ctx, "execute", "parse_error", recovered=False)
675
- return True
676
- ctx.trace.parse_error("execute", error=str(exc), recovered=True)
677
- self._emit_step(ctx, "execute", "parse_error", recovered=True)
678
- # Weak models often need one concrete reminder of the wire
679
- # format; feed it through the corrections channel and retry
680
- # instead of aborting the whole run on the first slip.
681
- hint = (
682
- 'Your last reply was not a single JSON action object. Reply with '
683
- 'EXACTLY one JSON object like {"thoughts": "...", "action": '
684
- '"tool_name", "args": {...}} and nothing else.'
685
- )
686
- if parse_failures >= profile.escalate_after:
687
- # Escalate: name the valid tools so the model stops
688
- # inventing action names or prose. The compact profile escalates
689
- # a slip earlier — a small model needs the list sooner.
690
- valid = ", ".join(sorted(self.deps.tool_governance.keys()))
691
- hint = (
692
- f"{hint} Valid action values are: {valid}, final. "
693
- 'Use {"action": "final", "message": "..."} to finish.'
694
- )
695
- if hint not in ctx.corrections:
696
- ctx.corrections.append(hint)
697
- ctx.trace.correction("execute", hint=hint)
698
- return False
699
-
700
- async def _direct_file_path(
701
- self, ctx: AgentRunContext, req: Any, current_user: str,
702
- model_id: Optional[str],
703
- ) -> bool:
704
- """Write the plan's file steps without asking the model for JSON (v9.9.7).
705
-
706
- The compact profile's escape hatch. A 1–4B local model that cannot hold
707
- the tool-call protocol can still write a file, so when JSON tool calls
708
- are exhausted the loop drops the protocol entirely: it takes the paths
709
- the *planner* already chose and asks only for file content in plain
710
- text, through the same validated
711
- :func:`~latticeai.core.file_generation.generate_file_content` pipeline
712
- the direct chat path uses.
713
-
714
- Returns True when at least one file was actually written. Honest
715
- failure modes: no planned paths, a governor that stages the write as a
716
- proposal, or a tool error all return False and leave the run to end as
717
- it would have — this never fabricates evidence.
718
- """
719
- d = self.deps
720
- planned: List[str] = []
721
- for step in ctx.plan.get("steps") or []:
722
- if not isinstance(step, dict) or step.get("action") not in d.file_create_actions:
723
- continue
724
- path = str((step.get("args") or {}).get("path") or "").strip()
725
- if path and path not in planned:
726
- planned.append(path)
727
- if not planned:
728
- inferred = infer_file_target(getattr(req, "message", "") or "")
729
- if inferred:
730
- planned = [inferred]
731
- if not planned:
732
- return False
733
-
734
- goal = str(ctx.plan.get("goal") or getattr(req, "message", "") or "")
735
-
736
- async def _generate(context: str) -> Any:
737
- return await d.generate_as(
738
- model_id,
739
- message="Write the file content.",
740
- context=context,
741
- max_tokens=self.phase_budgets.execute_tokens,
742
- temperature=0.2,
743
- )
744
-
745
- wrote = False
746
- for path in planned[:6]:
747
- try:
748
- content, meta = await generate_file_content(
749
- _generate,
750
- target_path=path,
751
- user_request=goal,
752
- bundle_files=planned if len(planned) > 1 else None,
753
- )
754
- except Exception as exc: # noqa: BLE001 — fallback must not raise
755
- logging.warning("direct file path generation failed for %s: %s", path, exc)
756
- continue
757
- ctx.trace.llm_call("execute", model=model_id)
758
- ctx.trace.repair("execute", repairs=["direct_path_fallback"])
759
- args = {"path": path, "content": content}
760
- policy = d.policy_for("write_file", args)
761
- risk = d.risk_level(policy)
762
- before = len(ctx.transcript)
763
- self._dispatch_step(ctx, "write_file", "direct path fallback", args, policy, risk, current_user)
764
- last = ctx.transcript[-1] if len(ctx.transcript) > before else {}
765
- if isinstance(last.get("result"), dict) and not last["result"].get("proposed"):
766
- wrote = True
767
- last["direct_path"] = True
768
- last["generation"] = {"repaired": bool(meta.get("repaired"))}
769
- if wrote:
770
- ctx.trace.decision("execute", decision="direct_path_fallback", files=len(planned))
771
- self._emit_step(ctx, "execute", "direct_path", files=len(planned))
772
- ctx.final_message = (
773
- "도구 호출 형식을 계속 벗어나서, 계획에 있던 파일을 직접 생성했습니다. "
774
- "내용을 확인해 주세요."
775
- )
776
- return wrote
777
-
778
- def _is_repeated_create(self, ctx: AgentRunContext, name: Any, args: dict) -> bool:
779
- """Loop guard: the same file-create action+args re-issued right after a result."""
780
- exec_steps = [s for s in ctx.transcript if s.get("state") == AgentState.EXECUTING.value]
781
- last = exec_steps[-1] if exec_steps else None
782
- return bool(
783
- name in self.deps.file_create_actions and last
784
- and last.get("action") == name
785
- and (last.get("args") or {}) == args
786
- and "result" in last
787
- )
788
-
789
- def _governor_review(
790
- self, ctx: AgentRunContext, name: str, thoughts: str, args: dict,
791
- policy: Mapping[str, Any], risk: str, current_user: str, request_workspace: Optional[str],
792
- conversation_id: Optional[str] = None,
793
- ) -> Tuple[bool, bool]:
794
- """Central change-class governance: create-new runs with minimal
795
- friction, change/delete-existing becomes a review proposal.
796
-
797
- Returns ``(proposed, governor_allows_additive)``: ``proposed`` means the
798
- step was staged as a proposal (skip execution); ``allows_additive`` lets
799
- an additive create pass the classic approval gate.
800
-
801
- Under a mode that does not stage proposals (``trusted`` / ``bypass``)
802
- the decision is made *before* the governor is consulted, because
803
- ``review`` persists a proposal as a side effect — reviewing first and
804
- discarding the verdict afterwards would apply the change *and* leave an
805
- orphan proposal pending in the Review Center.
806
- """
807
- d = self.deps
808
- if d.change_governor is None:
809
- return False, False
810
-
811
- mode = self.resolve_permission_mode(
812
- ctx, user_email=current_user, workspace_id=request_workspace,
813
- )
814
- if not should_stage_proposal(mode, proposal_required=True):
815
- if name not in self._governed_tools():
816
- return False, False
817
- if policy.get("destructive") or policy.get("risk") == "destructive":
818
- # Let the destructive gate downstream own the block + transcript.
819
- return False, False
820
- d.audit(
821
- "agent_change_auto_applied",
822
- user_email=current_user,
823
- workspace_id=request_workspace,
824
- action=name,
825
- path=str(args.get("path") or "") or None,
826
- permission_mode=mode.value,
827
- note="permission mode auto-applies mutation with audit",
828
- )
829
- return False, True
830
-
831
- verdict = d.change_governor.review(
832
- name, args, policy=dict(policy),
833
- user_email=current_user, workspace_id=request_workspace,
834
- conversation_id=conversation_id,
835
- )
836
- if verdict is not None and verdict.get("decision") == "proposed":
837
- proposal = verdict.get("proposal") or {}
838
- ctx.trace.tool("execute", name=name, outcome="proposed", risk=risk)
839
- self._emit_step(ctx, "execute", "proposed", action=name)
840
- ctx.transcript.append({
841
- "state": AgentState.EXECUTING.value, "action": name,
842
- "thoughts": thoughts, "args": {k: v for k, v in args.items() if k != "content"},
843
- "risk": risk, "governance": dict(policy),
844
- "result": {
845
- "proposed": True,
846
- "proposal_id": proposal.get("id"),
847
- "note": "기존 내용을 바꾸는 작업이라 변경 제안으로 저장했습니다. 검토함에서 승인하면 적용됩니다.",
848
- },
849
- })
850
- d.audit(
851
- "agent_change_proposed", user_email=current_user,
852
- action=name, proposal_id=proposal.get("id"),
853
- change_class=(verdict.get("classification") or {}).get("change_class"),
854
- )
855
- return True, False
856
- return False, (verdict is not None and verdict.get("decision") == "allow_additive")
857
-
858
- def _blocked_by_gates(
859
- self, ctx: AgentRunContext, req: Any, name: str, thoughts: str, args: dict,
860
- policy: Mapping[str, Any], risk: str, current_user: str, governor_allows_additive: bool,
861
- ) -> bool:
862
- """Destructive / circuit-breaker / fail-closed-overwrite / approval gates.
863
-
864
- Returns True when the step was blocked. The active permission mode can
865
- widen what runs without an extra approval prompt, but never widens a
866
- circuit breaker, the destructive gate, or the overwrite check.
867
- """
868
- d = self.deps
869
- mode = self.resolve_permission_mode(
870
- ctx,
871
- user_email=current_user,
872
- workspace_id=getattr(req, "workspace_id", None),
873
- )
874
- # Hard denials first — mode-invariant. A circuit breaker (root/home
875
- # paths, `rm -rf /` style commands) and a destructive policy are both
876
- # audited as ``blocked`` with the reason that actually fired, rather
877
- # than being flattened into the approval path.
878
- breaker = is_circuit_breaker(name, policy, args)
879
- hard_deny = breaker or (
880
- "destructive policy"
881
- if policy["risk"] == "destructive" or policy.get("destructive")
882
- else None
883
- )
884
- if hard_deny:
885
- error = (
886
- f"BLOCKED: destructive action '{name}' not permitted in agent mode."
887
- if hard_deny == "destructive policy"
888
- else f"BLOCKED: {hard_deny}"
889
- )
890
- ctx.trace.tool("execute", name=name, outcome="blocked_destructive", risk=risk)
891
- self._emit_step(ctx, "execute", "blocked", action=name, reason="destructive")
892
- ctx.transcript.append({
893
- "state": AgentState.EXECUTING.value, "action": name,
894
- "thoughts": thoughts, "args": args, "risk": risk,
895
- "governance": dict(policy),
896
- "permission_mode": mode.value,
897
- "error": error,
898
- })
899
- d.audit(
900
- "agent_blocked", user_email=current_user, source=getattr(req, "source", None) or "agent",
901
- action=name, reason="destructive", governance=dict(policy),
902
- )
903
- return True
904
-
905
- # Fail-closed overwrite guard — mode-invariant, like the two above.
906
- # A call that rewrites existing content but cannot be staged as a
907
- # reviewable proposal (binary document creators, home-sandbox writes)
908
- # has no safe apply path in ANY mode: trusted/bypass skip the approval
909
- # *prompt*, they never remove the existence check. Without this the
910
- # loop silently overwrote files that the HTTP surface refuses with 409
911
- # (``ToolDispatchService.enforce_policy``).
912
- overwrite = classify_tool_call(
913
- name, args, policy=dict(policy),
914
- path_exists=lambda candidate: self._governed_path_exists(name, candidate),
915
- )
916
- if overwrite.get("fail_closed"):
917
- target = str(args.get("path") or args.get("filename") or "")
918
- error = (
919
- f"NEEDS_REVIEW: '{name}' 은(는) 이미 있는 파일 '{target}' 을(를) 덮어씁니다. "
920
- "이 도구의 변경은 검토 가능한 제안으로 만들 수 없어 실행하지 않았습니다. "
921
- "새 파일 이름으로 만들거나 write_file/edit_file 로 수정하세요."
922
- )
923
- ctx.trace.tool("execute", name=name, outcome="blocked_overwrite", risk=risk)
924
- self._emit_step(ctx, "execute", "blocked", action=name, reason="overwrite")
925
- ctx.transcript.append({
926
- "state": AgentState.EXECUTING.value, "action": name,
927
- "thoughts": thoughts,
928
- # Same shape as a staged proposal: the payload is never worth
929
- # replaying into the transcript, only the decision is.
930
- "args": {k: v for k, v in args.items() if k != "content"},
931
- "risk": risk,
932
- "governance": dict(policy),
933
- "permission_mode": mode.value,
934
- "change_class": overwrite.get("change_class"),
935
- "error": error,
936
- })
937
- d.audit(
938
- "agent_blocked", user_email=current_user,
939
- source=getattr(req, "source", None) or "agent",
940
- action=name, reason="overwrite_fail_closed",
941
- path=target or None,
942
- change_class=overwrite.get("change_class"),
943
- permission_mode=mode.value,
944
- governance=dict(policy),
945
- )
946
- return True
947
-
948
- reason = block_reason_for_tool(
949
- mode, name, policy, args,
950
- approved_by_human=bool(ctx.approved_by_human),
951
- governor_allows_additive=governor_allows_additive,
952
- )
953
- if reason is None:
954
- return False
955
-
956
- d.audit(
957
- "agent_exec", user_email=current_user, source=getattr(req, "source", None) or "agent",
958
- state=AgentState.EXECUTING.value, action=name, risk=risk,
959
- shell=policy["shell"], network=policy["network"],
960
- destructive=policy["destructive"], sandbox=policy["sandbox"],
961
- rollback=policy["rollback"],
962
- permission_mode=mode.value,
963
- args={k: v for k, v in args.items() if k != "content"},
964
- )
965
- ctx.trace.tool("execute", name=name, outcome="blocked_approval", risk=risk)
966
- self._emit_step(ctx, "execute", "blocked", action=name, reason="approval")
967
- ctx.transcript.append({
968
- "state": AgentState.EXECUTING.value, "action": name,
969
- "thoughts": thoughts, "args": args, "risk": risk,
970
- "governance": dict(policy),
971
- "permission_mode": mode.value,
972
- "error": reason,
973
- })
974
- return True
975
-
976
- def _dispatch_step(
977
- self, ctx: AgentRunContext, name: str, thoughts: str, args: dict,
978
- policy: Mapping[str, Any], risk: str, current_user: str,
979
- ) -> None:
980
- """Role check + shared tool lifecycle, recorded on the transcript either way."""
981
- d = self.deps
982
- sanitize_meta: Optional[Dict[str, Any]] = None
983
- if name == "write_file" and isinstance(args.get("content"), str):
984
- # ArtifactWritePipeline: the executor's args.content is untrusted
985
- # model output. The same extract→validate→repair guarantee as the
986
- # direct chat path applies here, so a weak model driving the JSON
987
- # loop can never persist fenced/chatty/truncated payloads.
988
- cleaned, meta = sanitize_write_content(
989
- str(args.get("path") or ""), args["content"],
990
- user_request=str(ctx.plan.get("goal") or thoughts or name),
991
- )
992
- if meta.get("sanitized"):
993
- args = dict(args)
994
- args["content"] = cleaned
995
- sanitize_meta = meta
996
- ctx.trace.repair(
997
- "execute",
998
- repairs=[
999
- "artifact_repair" if meta.get("repaired") else "artifact_sanitize"
1000
- ],
1001
- )
1002
- step_index = 1 + sum(
1003
- 1 for s in ctx.transcript
1004
- if s.get("state") == AgentState.EXECUTING.value
1005
- and s.get("action") not in (None, "final", "parse_error")
1006
- )
1007
- if (
1008
- name in d.file_create_actions
1009
- and d.snapshot_file is not None
1010
- and args.get("path")
1011
- ):
1012
- # Pre-write snapshot (review L7): the first capture per path is
1013
- # the true pre-run state — later writes to the same path must
1014
- # not overwrite it. Best-effort: a snapshot failure never
1015
- # blocks the write, it only narrows rollback options.
1016
- path_str = str(args["path"])
1017
- if not any(entry.get("path") == path_str for entry in ctx.rollback_log):
1018
- try:
1019
- pre = d.snapshot_file(path_str)
1020
- ctx.rollback_log.append({"path": path_str, **(pre or {})})
1021
- except Exception as exc: # noqa: BLE001
1022
- logging.warning("pre-write snapshot failed for %s: %s", path_str, exc)
1023
- try:
1024
- d.check_role(name, current_user)
1025
- # Shared tool lifecycle: pre_tool (may block) → execute → post_tool.
1026
- result = dispatch_tool(
1027
- d.hooks, name, args,
1028
- lambda: d.execute_tool(name, args),
1029
- user_email=current_user, source="agent",
1030
- )
1031
- ctx.trace.tool("execute", name=name, outcome="ok", risk=risk)
1032
- ctx.transcript.append({
1033
- "state": AgentState.EXECUTING.value, "action": name,
1034
- "thoughts": thoughts, "args": args,
1035
- "risk": risk, "governance": dict(policy), "result": result,
1036
- **({"content_sanitize": sanitize_meta} if sanitize_meta else {}),
1037
- })
1038
- self._emit_step(
1039
- ctx, "execute", "tool", action=name, ok=True, step=step_index,
1040
- path=str(args.get("path")) if args.get("path") else None,
1041
- )
1042
- except (ToolError, KeyError, TypeError, PermissionError) as exc:
1043
- ctx.trace.tool("execute", name=name, outcome="error", risk=risk)
1044
- ctx.transcript.append({
1045
- "state": AgentState.EXECUTING.value, "action": name,
1046
- "thoughts": thoughts, "args": args,
1047
- "risk": risk, "governance": dict(policy), "error": str(exc),
1048
- })
1049
- self._emit_step(
1050
- ctx, "execute", "tool", action=name, ok=False, step=step_index,
1051
- path=str(args.get("path")) if args.get("path") else None,
1052
- )
1053
-
1054
- # ── VERIFY ───────────────────────────────────────────────────────
1055
- def _has_execution_evidence(self, ctx: AgentRunContext) -> bool:
1056
- """Deterministic evidence check: at least one executing step actually
1057
- produced a result (tool ran, or a governed change was staged as a
1058
- proposal). ``final``/parse-error/blocked steps carry no result and do
1059
- not count — a critic PASS over an evidence-free transcript must not
1060
- become DONE."""
1061
- for step in ctx.transcript:
1062
- if step.get("state") != AgentState.EXECUTING.value:
1063
- continue
1064
- if step.get("action") in (None, "final", "parse_error"):
1065
- continue
1066
- if isinstance(step.get("result"), dict):
1067
- return True
1068
- return False
1069
-
1070
- async def verify(
1071
- self, ctx: AgentRunContext, req: Any, lang_hint: str, current_user: str,
1072
- max_retry: int = 3, model_id: Optional[str] = None,
1073
- ) -> None:
1074
- """VERIFYING: Critic role evaluates transcript → DONE / EXECUTING (retry) / ROLLBACK / NEEDS_REVIEW / FAILED.
1075
-
1076
- Fail-closed: a critic whose output cannot be parsed (after one strict
1077
- repair retry) never fabricates a PASS — the run terminates as
1078
- NEEDS_REVIEW so the user is told to check the result themselves.
1079
- """
1080
- d = self.deps
1081
- # The critic must see every step (evidence completeness), but not
1082
- # every byte of tool output — long bodies are capped per string so
1083
- # verification stays affordable on long runs (review Wave 0.3).
1084
- verify_transcript = _truncate_strings(
1085
- ctx.transcript, self.transcript_budget.verify_chars
1086
- )
1087
- # Deterministic artifact facts (review L4): the critic sees the
1088
- # sanitize/repair honesty flags per written file, not just prose.
1089
- checklist = artifact_checklist(ctx.transcript, d.file_create_actions)
1090
- checklist_hint = (
1091
- f"\n\n{format_artifact_checklist(checklist)}" if checklist else ""
1092
- )
1093
- # Requirement coverage (review 루프 §2): the critic previously judged
1094
- # "did this fulfill the request?" from prose alone. It now also sees
1095
- # which requested files actually exist and which requirements the user
1096
- # spelled out.
1097
- coverage = requirement_coverage(
1098
- req.message, ctx.transcript, d.file_create_actions
1099
- )
1100
- context = (
1101
- f"{d.critic_prompt}\n\n"
1102
- f"[LANGUAGE HINT: {lang_hint}]\n\n"
1103
- f"Original request: {req.message}\n"
1104
- f"Plan goal: {ctx.plan.get('goal', req.message)}{checklist_hint}"
1105
- f"{format_requirement_coverage(coverage)}\n\n"
1106
- f"Full transcript:\n{json.dumps(verify_transcript, ensure_ascii=False, indent=2)}"
1107
- )
1108
- raw = await d.generate_as(
1109
- model_id,
1110
- message="Review the execution transcript and return your verdict JSON.",
1111
- context=context, max_tokens=self.phase_budgets.verify_tokens, temperature=0.1,
1112
- )
1113
- ctx.trace.llm_call("verify", model=model_id)
1114
- verdict: Optional[Dict[str, Any]] = None
1115
- try:
1116
- verdict, verdict_repairs = extract_action_details(str(raw))
1117
- ctx.trace.repair("verify", repairs=verdict_repairs)
1118
- except ValueError as exc:
1119
- # One strict repair retry — re-ask the critic for the exact wire
1120
- # format instead of fabricating a verdict.
1121
- ctx.trace.parse_error("verify", error=str(exc), recovered=True)
1122
- strict_context = (
1123
- f"{context}\n\n"
1124
- "Your previous verdict was not parseable JSON. Reply with EXACTLY one "
1125
- 'JSON object like {"action": "verdict", "verdict": "PASS", '
1126
- '"next_state": "DONE", "reason": "...", "corrections": []} '
1127
- "and nothing else. verdict must be PASS or FAIL; next_state must be "
1128
- "one of DONE, EXECUTING, ROLLBACK, FAILED."
1129
- )
1130
- raw = await d.generate_as(
1131
- model_id,
1132
- message="Return your verdict as one strict JSON object.",
1133
- context=strict_context, max_tokens=self.phase_budgets.verify_tokens,
1134
- temperature=0.0,
1135
- )
1136
- ctx.trace.llm_call("verify", model=model_id)
1137
- try:
1138
- verdict, verdict_repairs = extract_action_details(str(raw))
1139
- ctx.trace.repair("verify", repairs=verdict_repairs)
1140
- except ValueError as retry_exc:
1141
- ctx.trace.parse_error("verify", error=str(retry_exc), recovered=False)
1142
- verdict = None
1143
-
1144
- has_evidence = self._has_execution_evidence(ctx)
1145
-
1146
- if verdict is None:
1147
- # Verifier unavailable — fail closed, never DONE.
1148
- ctx.transcript.append({
1149
- "state": AgentState.VERIFYING.value,
1150
- "verdict": "UNAVAILABLE",
1151
- "reason": "critic output unparseable after strict retry",
1152
- "verifier_available": False,
1153
- "verdict_valid": False,
1154
- "evidence": has_evidence,
1155
- })
1156
- ctx.trace.decision(
1157
- "verify", decision="verification_unavailable",
1158
- verifier_available=False, verdict_valid=False, evidence=has_evidence,
1159
- )
1160
- self._emit_step(ctx, "verify", "verdict", verdict="UNAVAILABLE")
1161
- ctx.final_message = (
1162
- "검증을 완료하지 못했습니다 — 검증 모델의 응답을 해석할 수 없었습니다. "
1163
- "실행 결과를 직접 확인해 주시고, 필요하면 다시 시도해 주세요."
1164
- )
1165
- ctx.state = AgentState.NEEDS_REVIEW
1166
- return
1167
-
1168
- ctx.corrections = verdict.get("corrections", [])
1169
- # Normalize legacy verdict next_state strings to current AgentState names
1170
- raw_next = verdict.get("next_state", "")
1171
- next_s = {"COMPLETE": "DONE", "RETRY": "EXECUTING"}.get(raw_next, raw_next)
1172
-
1173
- ctx.transcript.append({
1174
- "state": AgentState.VERIFYING.value,
1175
- "verdict": verdict.get("verdict", ""),
1176
- "reason": verdict.get("reason", ""),
1177
- "corrections": ctx.corrections,
1178
- "confidence": verdict.get("confidence", 0.9),
1179
- "next_state": next_s,
1180
- "verifier_available": True,
1181
- "verdict_valid": True,
1182
- "evidence": has_evidence,
1183
- })
1184
-
1185
- ctx.trace.decision(
1186
- "verify", decision=str(verdict.get("verdict", "")), next_state=next_s,
1187
- verifier_available=True, verdict_valid=True, evidence=has_evidence,
1188
- )
1189
- self._emit_step(
1190
- ctx, "verify", "verdict",
1191
- verdict=str(verdict.get("verdict", "")), next_state=next_s,
1192
- )
1193
- if verdict.get("verdict") == "PASS":
1194
- # DONE requires both: a validly parsed PASS verdict AND
1195
- # deterministic execution evidence in the transcript. A PASS over
1196
- # an evidence-free run is not a completion.
1197
- if not has_evidence:
1198
- ctx.trace.decision("verify", decision="needs_review_no_evidence")
1199
- ctx.final_message = (
1200
- "검증자는 통과를 보고했지만 실제 실행 근거(도구 실행 기록)가 없어 "
1201
- "완료로 처리하지 않았습니다. 결과를 직접 확인해 주세요."
1202
- )
1203
- ctx.state = AgentState.NEEDS_REVIEW
1204
- return
1205
- if not coverage["complete"]:
1206
- # A PASS that leaves a *requested file* unwritten is not a
1207
- # completion — this is a fact, not a judgement, so it is
1208
- # enforced rather than merely reported to the critic.
1209
- missing = ", ".join(coverage["missing_files"])
1210
- ctx.trace.decision(
1211
- "verify", decision="needs_review_missing_files",
1212
- missing=len(coverage["missing_files"]),
1213
- )
1214
- ctx.transcript.append({
1215
- "state": AgentState.VERIFYING.value,
1216
- "requirement_coverage": coverage,
1217
- })
1218
- ctx.final_message = (
1219
- f"요청한 파일 중 일부가 만들어지지 않아 완료로 처리하지 않았습니다: {missing}"
1220
- )
1221
- ctx.state = AgentState.NEEDS_REVIEW
1222
- return
1223
- if not ctx.final_message:
1224
- ctx.final_message = verdict.get("reason", "작업이 완료되었습니다.")
1225
- ctx.state = AgentState.DONE
1226
- elif next_s == "ROLLBACK":
1227
- ctx.state = AgentState.ROLLBACK
1228
- elif next_s == "EXECUTING":
1229
- if ctx.retry_count >= max_retry:
1230
- ctx.final_message = "처리 중 문제가 발생했습니다. 다시 시도해 주세요."
1231
- ctx.state = AgentState.FAILED
1232
- else:
1233
- ctx.retry_count += 1
1234
- ctx.trace.retry("verify", attempt=ctx.retry_count)
1235
- ctx.transcript.append({
1236
- "state": AgentState.EXECUTING.value,
1237
- "retry_attempt": ctx.retry_count,
1238
- "corrections": ctx.corrections,
1239
- })
1240
- ctx.state = AgentState.EXECUTING
1241
- elif next_s == "DONE":
1242
- # Contradictory verdict: the critic asked for DONE without a PASS.
1243
- # The loose "or next_state == DONE" success path is gone — this is
1244
- # a non-success that the user must review.
1245
- ctx.trace.decision("verify", decision="needs_review_inconsistent_verdict")
1246
- ctx.final_message = (
1247
- "검증 결과가 일관되지 않아 완료로 처리하지 않았습니다. "
1248
- "실행 결과를 직접 확인해 주세요."
1249
- )
1250
- ctx.state = AgentState.NEEDS_REVIEW
1251
- else:
1252
- ctx.final_message = verdict.get("reason", "검증자가 인식되지 않은 다음 상태를 반환했습니다.")
1253
- ctx.state = AgentState.FAILED
1254
-
1255
- # ── ROLLBACK ─────────────────────────────────────────────────────
1256
- def _snapshot_for(self, ctx: AgentRunContext, path: str) -> Optional[Dict[str, Any]]:
1257
- for entry in ctx.rollback_log:
1258
- if entry.get("path") == path:
1259
- return entry
1260
- return None
1261
-
1262
- def _rollback_one(self, ctx: AgentRunContext, path: str, gov: Dict[str, Any]) -> Dict[str, Any]:
1263
- """Recover one path: git when governed and available, else the
1264
- pre-write snapshot, else an honest ``mode="none"`` (review L7)."""
1265
- d = self.deps
1266
- if gov.get("rollback") == "git" and d.rollback_file is not None:
1267
- try:
1268
- result = dict(d.rollback_file(str(path)))
1269
- except Exception as exc: # noqa: BLE001
1270
- result = {"path": path, "ok": False, "error": str(exc)}
1271
- if result.get("ok"):
1272
- result["mode"] = "git"
1273
- return result
1274
- snapshot = self._snapshot_for(ctx, str(path))
1275
- if snapshot is not None and d.restore_snapshot is not None and not snapshot.get("too_large"):
1276
- content = snapshot.get("content") if snapshot.get("existed") else None
1277
- try:
1278
- restored = dict(d.restore_snapshot(str(path), content))
1279
- except Exception as exc: # noqa: BLE001
1280
- restored = {"path": path, "ok": False, "error": str(exc)}
1281
- restored.setdefault("path", path)
1282
- restored["mode"] = "snapshot"
1283
- return restored
1284
- return {
1285
- "path": path, "ok": False, "mode": "none",
1286
- "error": "no rollback available (git not applicable, no usable snapshot)",
1287
- }
1288
-
1289
- def rollback(self, ctx: AgentRunContext, current_user: str) -> None:
1290
- """ROLLBACK: recover written files (git → snapshot → none), then FAILED."""
1291
- d = self.deps
1292
- rolled: List[dict] = []
1293
- seen_paths: set = set()
1294
- for step in ctx.transcript:
1295
- if step.get("state") != AgentState.EXECUTING.value:
1296
- continue
1297
- if not isinstance(step.get("result"), dict):
1298
- continue
1299
- gov = step.get("governance", {}) or {}
1300
- path = step["result"].get("path") or (step.get("args") or {}).get("path", "")
1301
- if not path or str(path) in seen_paths:
1302
- continue
1303
- if gov.get("rollback") != "git" and step.get("action") not in d.file_create_actions:
1304
- continue
1305
- seen_paths.add(str(path))
1306
- rolled.append(self._rollback_one(ctx, str(path), gov))
1307
-
1308
- ctx.transcript.append({"state": AgentState.ROLLBACK.value, "rolled_back": rolled})
1309
- ctx.trace.decision(
1310
- "rollback", decision="rolled_back",
1311
- attempted=len(rolled), recovered=sum(1 for r in rolled if r.get("ok")),
1312
- )
1313
- recovered = [f"{r['path']} ({r.get('mode')})" for r in rolled if r.get("ok")]
1314
- ctx.final_message = (
1315
- f"실행 실패로 롤백했습니다. 복구 파일: {recovered}"
1316
- if recovered
1317
- else "롤백을 시도했으나 복구할 파일이 없거나 git/스냅샷 복구 수단이 없습니다."
1318
- )
1319
- d.audit("agent_rollback", user_email=current_user, rolled_back=rolled)
1320
- self._emit_step(ctx, "rollback", "rolled_back", recovered=len(recovered))
1321
- # Rollback is a recovery from a failed verification — terminal state is FAILED
1322
- ctx.state = AgentState.FAILED
1323
-
1324
- # ── MEMORY ───────────────────────────────────────────────────────
1325
- async def memory_update(self, ctx: AgentRunContext, req: Any, current_user: str) -> None:
1326
- """Background: Memory Updater role extracts learnings from a terminal run.
1327
-
1328
- Terminal-state learning policy (review §4.2 L6): DONE runs record what
1329
- worked; FAILED / NEEDS_REVIEW runs record what went wrong — failure is
1330
- exactly the experience worth remembering. The run status stored with
1331
- the experience is the *actual* terminal state, never a blanket "ok".
1332
- """
1333
- d = self.deps
1334
- terminal = ctx.state.value if ctx.state in AGENT_TERMINAL_STATES else "UNKNOWN"
1335
- outcome_hint = (
1336
- "The task completed successfully."
1337
- if ctx.state == AgentState.DONE
1338
- else (
1339
- f"The task ended as {terminal} — extract what went wrong and "
1340
- "what to do differently next time, not a success story."
1341
- )
1342
- )
1343
- context = (
1344
- f"{d.memory_updater_prompt}\n\n"
1345
- f"Task: {req.message}\n"
1346
- f"Terminal status: {terminal}. {outcome_hint}\n\n"
1347
- f"Last 5 transcript steps:\n{json.dumps(ctx.transcript[-5:], ensure_ascii=False)}"
1348
- )
1349
- try:
1350
- raw = await d.generate(
1351
- message="Extract learnings from this completed task.",
1352
- context=context, max_tokens=self.phase_budgets.memory_tokens, temperature=0.1,
1353
- )
1354
- mem = extract_action(str(raw))
1355
- kept_learnings = filter_learnings(mem.get("learnings") or [])
1356
- if mem.get("save_to_knowledge") and kept_learnings:
1357
- learnings = "\n".join(kept_learnings)
1358
- status_label = {
1359
- AgentState.DONE: "ok",
1360
- AgentState.NEEDS_REVIEW: "needs_review",
1361
- AgentState.FAILED: "failed",
1362
- }.get(ctx.state, "unknown")
1363
- if d.brain_memory is not None:
1364
- # This runtime is LLM-driven — its learnings are real
1365
- # experiences and enter the brain with provenance.
1366
- d.brain_memory.record_experience(
1367
- f"Agent: {req.message[:60]}",
1368
- learnings,
1369
- run={
1370
- "mode": "llm",
1371
- "status": status_label,
1372
- "agent_id": "agent:executor",
1373
- "steps": len(ctx.transcript),
1374
- },
1375
- user_email=current_user or None,
1376
- )
1377
- else:
1378
- d.knowledge_save(
1379
- learnings,
1380
- folder="30_Projects",
1381
- title=f"Agent: {req.message[:60]}",
1382
- )
1383
- except Exception as exc:
1384
- # Never crash a completed run, but never swallow silently either.
1385
- logging.warning("agent memory update failed: %s", exc)
1386
-
1387
- # ── DRIVE LOOP ───────────────────────────────────────────────────
1388
- async def run_to_completion(
1389
- self, ctx: AgentRunContext, req: Any, lang_hint: str,
1390
- current_user: str, max_steps: int, max_retry: int,
1391
- ) -> None:
1392
- """Run EXECUTING → VERIFYING → ROLLBACK loop until a terminal state."""
1393
- while ctx.state not in AGENT_TERMINAL_STATES:
1394
- ctx.state_history.append(ctx.state.value)
1395
- if len(ctx.state_history) > 200:
1396
- ctx.final_message = "에이전트 상태 머신이 최대 반복(200)에 도달해 중단했습니다."
1397
- ctx.state = AgentState.FAILED
1398
- break
1399
-
1400
- if ctx.state == AgentState.EXECUTING:
1401
- await self.execute(ctx, req, lang_hint, current_user, max_steps,
1402
- model_id=ctx.executing_model)
1403
- elif ctx.state == AgentState.VERIFYING:
1404
- await self.verify(ctx, req, lang_hint, current_user, max_retry,
1405
- model_id=ctx.reviewing_model)
1406
- elif ctx.state == AgentState.ROLLBACK:
1407
- self.rollback(ctx, current_user)
1408
- else:
1409
- ctx.state = AgentState.FAILED
1410
-
1411
- ctx.state_history.append(ctx.state.value)
1412
- self._emit_step(ctx, "terminal", "state", state=ctx.state.value)