ltcai 11.2.0 → 11.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -53
- package/docs/CHANGELOG.md +61 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +9 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +407 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/generate_rust_parity_fixtures.py +562 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +31 -2
- package/src-tauri/Cargo.lock +361 -3
- package/src-tauri/Cargo.toml +6 -1
- package/src-tauri/src/backend.rs +349 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +97 -399
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-yYpYnn0v.js +1 -0
- package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
- package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
- package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
- package/static/app/assets/Capture-CFIRsFNE.js +1 -0
- package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
- package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
- package/static/app/assets/Library-DwO3yZST.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
- package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
- package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
- package/static/app/assets/System-DW8F-2xL.js +1 -0
- package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
- package/static/app/assets/brain-Ci1CkWjM.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
- package/static/app/assets/circle-check-DfInj-qD.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-_u5iUHDr.js +10 -0
- package/static/app/assets/input-B0lPdRQZ.js +1 -0
- package/static/app/assets/link-2-CoFbooHS.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
- package/static/app/assets/primitives-DEbN-d6p.js +1 -0
- package/static/app/assets/search-BybIWPNd.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
- package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
- package/static/app/assets/utils-BlZr7Pd4.js +4 -0
- package/static/app/assets/workspace-jJY4RuAV.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
package/latticeai/core/agent.py
DELETED
|
@@ -1,1465 +0,0 @@
|
|
|
1
|
-
"""Single-agent runtime — the Discover→Plan→Implement→Verify state machine.
|
|
2
|
-
|
|
3
|
-
This module is the deep single-agent loop: a small interface (``AgentDeps`` ports +
|
|
4
|
-
``SingleAgentRuntime.run_to_completion``) over the whole role-phased state machine
|
|
5
|
-
(planner → executor → critic → rollback → memory). It carries no FastAPI,
|
|
6
|
-
no globals, and no I/O of its own — every collaborator is injected through
|
|
7
|
-
``AgentDeps``.
|
|
8
|
-
|
|
9
|
-
Two adapters justify the seam:
|
|
10
|
-
|
|
11
|
-
* production wires ``AgentDeps`` from ``latticeai.server_app``'s ``LLMRouter``, governance
|
|
12
|
-
map, audit log, and prompts;
|
|
13
|
-
* tests pass fake ports (an LLM that returns canned JSON, a recording tool
|
|
14
|
-
executor) and drive a full PLAN→EXECUTE→VERIFY→DONE cycle without a server.
|
|
15
|
-
|
|
16
|
-
HTTP concerns — request parsing, chat-history persistence, response shaping,
|
|
17
|
-
scheduling the background memory update — stay in the app layer. This module
|
|
18
|
-
only owns the state machine.
|
|
19
|
-
"""
|
|
20
|
-
|
|
21
|
-
from __future__ import annotations
|
|
22
|
-
|
|
23
|
-
import json
|
|
24
|
-
import logging
|
|
25
|
-
from dataclasses import dataclass
|
|
26
|
-
from pathlib import Path
|
|
27
|
-
from typing import (
|
|
28
|
-
Any,
|
|
29
|
-
Awaitable,
|
|
30
|
-
Callable,
|
|
31
|
-
Dict,
|
|
32
|
-
FrozenSet,
|
|
33
|
-
List,
|
|
34
|
-
Mapping,
|
|
35
|
-
Optional,
|
|
36
|
-
Tuple,
|
|
37
|
-
)
|
|
38
|
-
|
|
39
|
-
from lattice_brain.runtime.contracts import (
|
|
40
|
-
runtime_boundary_contract,
|
|
41
|
-
single_agent_contract,
|
|
42
|
-
)
|
|
43
|
-
from lattice_brain.runtime.hooks import dispatch_tool
|
|
44
|
-
from latticeai.core.agent_helpers import (
|
|
45
|
-
PhaseBudgets,
|
|
46
|
-
TranscriptBudget,
|
|
47
|
-
_truncate_strings,
|
|
48
|
-
artifact_checklist,
|
|
49
|
-
compact_transcript,
|
|
50
|
-
extract_action,
|
|
51
|
-
extract_action_details,
|
|
52
|
-
files_written,
|
|
53
|
-
filter_learnings,
|
|
54
|
-
format_artifact_checklist,
|
|
55
|
-
format_requirement_coverage,
|
|
56
|
-
normalize_plan,
|
|
57
|
-
requirement_coverage,
|
|
58
|
-
)
|
|
59
|
-
from latticeai.core.agent_permission import (
|
|
60
|
-
block_reason_for_tool,
|
|
61
|
-
non_auto_plan_steps,
|
|
62
|
-
resolve_deps_mode,
|
|
63
|
-
)
|
|
64
|
-
from latticeai.core.agent_profiles import AgentProfile, profile_for_model
|
|
65
|
-
from latticeai.core.agent_prompts import executor_prompt_for
|
|
66
|
-
|
|
67
|
-
# The state vocabulary and the pure helpers live in sibling modules so this one
|
|
68
|
-
# holds only the loop. They are re-exported (see ``__all__``) because callers —
|
|
69
|
-
# the HTTP layer, run_store, the eval harness, and the tests — have always
|
|
70
|
-
# imported them from here, and that contract does not change.
|
|
71
|
-
from latticeai.core.agent_state import AGENT_TERMINAL_STATES, AgentState
|
|
72
|
-
from latticeai.core.agent_trace import LoopTrace
|
|
73
|
-
from latticeai.core.file_generation import (
|
|
74
|
-
generate_file_content,
|
|
75
|
-
infer_file_target,
|
|
76
|
-
sanitize_write_content,
|
|
77
|
-
)
|
|
78
|
-
from latticeai.core.permission_mode import (
|
|
79
|
-
PermissionMode,
|
|
80
|
-
is_circuit_breaker,
|
|
81
|
-
plan_requires_approval,
|
|
82
|
-
should_stage_proposal,
|
|
83
|
-
)
|
|
84
|
-
from latticeai.core.tool_governor import classify_tool_call
|
|
85
|
-
from latticeai.core.tool_registry import SCOPED_KNOWLEDGE_TOOLS
|
|
86
|
-
from latticeai.tools import ToolError, document_output_target
|
|
87
|
-
|
|
88
|
-
__all__ = [
|
|
89
|
-
# this module
|
|
90
|
-
"AgentDeps",
|
|
91
|
-
"AgentRunContext",
|
|
92
|
-
"SingleAgentRuntime",
|
|
93
|
-
# re-exported from agent_state
|
|
94
|
-
"AGENT_TERMINAL_STATES",
|
|
95
|
-
"AgentState",
|
|
96
|
-
# re-exported from agent_helpers
|
|
97
|
-
"PhaseBudgets",
|
|
98
|
-
"TranscriptBudget",
|
|
99
|
-
"artifact_checklist",
|
|
100
|
-
"compact_transcript",
|
|
101
|
-
"extract_action",
|
|
102
|
-
"extract_action_details",
|
|
103
|
-
"files_written",
|
|
104
|
-
"filter_learnings",
|
|
105
|
-
"format_artifact_checklist",
|
|
106
|
-
"format_requirement_coverage",
|
|
107
|
-
"normalize_plan",
|
|
108
|
-
"requirement_coverage",
|
|
109
|
-
]
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
class AgentRunContext:
|
|
113
|
-
"""Mutable state carrier passed through all agent phases."""
|
|
114
|
-
__slots__ = ("state", "plan", "transcript", "retry_count",
|
|
115
|
-
"state_history", "corrections", "final_message", "rollback_log",
|
|
116
|
-
"executing_model", "reviewing_model", "approved_by_human", "trace",
|
|
117
|
-
"on_step", "project_context", "permission_mode",
|
|
118
|
-
"self_model_summary")
|
|
119
|
-
|
|
120
|
-
def __init__(self) -> None:
|
|
121
|
-
self.state: AgentState = AgentState.IDLE
|
|
122
|
-
self.trace: LoopTrace = LoopTrace()
|
|
123
|
-
self.plan: dict = {}
|
|
124
|
-
self.transcript: list = []
|
|
125
|
-
self.retry_count: int = 0
|
|
126
|
-
self.state_history: list = []
|
|
127
|
-
self.corrections: list = []
|
|
128
|
-
self.final_message: str = ""
|
|
129
|
-
self.rollback_log: list = []
|
|
130
|
-
self.executing_model: Optional[str] = None
|
|
131
|
-
self.reviewing_model: Optional[str] = None
|
|
132
|
-
self.approved_by_human: bool = False
|
|
133
|
-
# Per-run step observer (review Wave 1.1): the HTTP layer attaches a
|
|
134
|
-
# callback here so live SSE clients see progress while EXECUTING.
|
|
135
|
-
# Never serialized; a broken observer never breaks the loop.
|
|
136
|
-
self.on_step: Optional[Callable[[Dict[str, Any]], None]] = None
|
|
137
|
-
# Multi-turn project loop (v9.9.6): a prompt block describing where the
|
|
138
|
-
# project stands — files already produced, open TODOs, the last honest
|
|
139
|
-
# verification. Empty for a standalone run, which behaves exactly as
|
|
140
|
-
# before. Set by the HTTP layer, read by plan/execute/verify.
|
|
141
|
-
self.project_context: str = ""
|
|
142
|
-
# Autonomy dial resolved once per run (v9.9.8). The HTTP layer stamps
|
|
143
|
-
# the user/workspace-scoped mode here so the plan gate and every
|
|
144
|
-
# per-tool gate in the same run agree; ``None`` falls back to the
|
|
145
|
-
# process-wide resolver on ``deps``.
|
|
146
|
-
self.permission_mode: Optional[str] = None
|
|
147
|
-
# Self-Model summary resolved once per run (v11.2.0). The executor
|
|
148
|
-
# prompt is rebuilt on every turn of the loop; reading the profile
|
|
149
|
-
# once and reusing it keeps a graph query off that hot path — and
|
|
150
|
-
# keeps every turn of one run describing the same person. ``None``
|
|
151
|
-
# means "not resolved yet"; ``""`` is a resolved, empty profile.
|
|
152
|
-
self.self_model_summary: Optional[str] = None
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
@dataclass
|
|
156
|
-
class AgentDeps:
|
|
157
|
-
"""The ports a :class:`SingleAgentRuntime` needs from the outside world.
|
|
158
|
-
|
|
159
|
-
Everything the state machine touches is here, so the loop can be exercised
|
|
160
|
-
against fakes. See module docstring for the two-adapter rationale.
|
|
161
|
-
"""
|
|
162
|
-
|
|
163
|
-
# ── LLM port ─────────────────────────────────────────────────────
|
|
164
|
-
# generate_as(model_id, message, context, max_tokens, temperature) -> str
|
|
165
|
-
generate_as: Callable[..., Awaitable[Any]]
|
|
166
|
-
# generate(message, context, max_tokens, temperature) -> str
|
|
167
|
-
generate: Callable[..., Awaitable[Any]]
|
|
168
|
-
|
|
169
|
-
# ── tool port ────────────────────────────────────────────────────
|
|
170
|
-
execute_tool: Callable[[str, dict], dict]
|
|
171
|
-
policy_for: Callable[[str, dict], Mapping[str, Any]] # name, args -> policy
|
|
172
|
-
risk_level: Callable[[Any], str] # policy -> "low"|"medium"|"high"
|
|
173
|
-
check_role: Callable[[str, str], None] # tool_name, user -> raises if not allowed
|
|
174
|
-
tool_governance: Mapping[str, Mapping[str, Any]] # name -> policy (auto_approve set)
|
|
175
|
-
file_create_actions: FrozenSet[str]
|
|
176
|
-
|
|
177
|
-
# ── context / memory / audit ports ───────────────────────────────
|
|
178
|
-
recent_chat_context: Callable[..., str] # (conversation_id=...) -> str
|
|
179
|
-
clear_history: Callable[[int], dict]
|
|
180
|
-
knowledge_save: Callable[..., Any]
|
|
181
|
-
audit: Callable[..., None] # (event, **kw) -> None
|
|
182
|
-
|
|
183
|
-
# ── prompts + config ─────────────────────────────────────────────
|
|
184
|
-
planner_prompt: str
|
|
185
|
-
executor_prompt: str
|
|
186
|
-
critic_prompt: str
|
|
187
|
-
memory_updater_prompt: str
|
|
188
|
-
agent_root: Path
|
|
189
|
-
|
|
190
|
-
# ── rollback port (optional) ─────────────────────────────────────
|
|
191
|
-
# Production injects this from the tool dispatch service so this pure
|
|
192
|
-
# state machine does not shell out directly. Tests can pass a recorder.
|
|
193
|
-
rollback_file: Optional[Callable[[str], Dict[str, Any]]] = None
|
|
194
|
-
|
|
195
|
-
# ── snapshot rollback ports (optional, review L7) ────────────────
|
|
196
|
-
# git-only rollback left non-git workspaces and newly created files
|
|
197
|
-
# unrecoverable. ``snapshot_file(path)`` captures pre-write state
|
|
198
|
-
# ({"existed", "content", "too_large"}) before a file-create action;
|
|
199
|
-
# ``restore_snapshot(path, content)`` restores it (content=None deletes
|
|
200
|
-
# a file the run created). Both are production-wired with workspace
|
|
201
|
-
# path safety; tests pass recorders.
|
|
202
|
-
snapshot_file: Optional[Callable[[str], Dict[str, Any]]] = None
|
|
203
|
-
restore_snapshot: Optional[Callable[[str, Optional[str]], Dict[str, Any]]] = None
|
|
204
|
-
|
|
205
|
-
# ── lifecycle hooks port (optional) ──────────────────────────────
|
|
206
|
-
# When present, every tool execution fires the shared pre_tool/post_tool
|
|
207
|
-
# lifecycle, so the agent tool path no longer bypasses hooks.
|
|
208
|
-
hooks: Any = None
|
|
209
|
-
|
|
210
|
-
# ── brain memory port (optional) ─────────────────────────────────
|
|
211
|
-
# When present, completed-run learnings become typed Experience records
|
|
212
|
-
# through the unified ingestion pipeline (with provenance), replacing
|
|
213
|
-
# the vault markdown dump.
|
|
214
|
-
brain_memory: Any = None
|
|
215
|
-
|
|
216
|
-
# ── change governor port (optional) ──────────────────────────────
|
|
217
|
-
# When present, file writes are classified centrally: additive creates
|
|
218
|
-
# run with minimal friction, while mutations/deletions of existing
|
|
219
|
-
# content are staged as review proposals instead of applied. The port is
|
|
220
|
-
# ``review(name, args, policy=..., user_email=..., workspace_id=...)``
|
|
221
|
-
# returning None (fall through to the classic gates) or a verdict dict.
|
|
222
|
-
change_governor: Any = None
|
|
223
|
-
|
|
224
|
-
# ── phase budgets (optional) ─────────────────────────────────────
|
|
225
|
-
# Per-phase token caps (plan/execute/verify/memory). None reads the
|
|
226
|
-
# environment once at first use; tests inject a fixed PhaseBudgets.
|
|
227
|
-
phase_budgets: Optional[PhaseBudgets] = None
|
|
228
|
-
|
|
229
|
-
# ── transcript shaping (optional) ────────────────────────────────
|
|
230
|
-
# Executor/critic prompt window caps. None reads the environment once;
|
|
231
|
-
# tests inject a fixed TranscriptBudget.
|
|
232
|
-
transcript_budget: Optional["TranscriptBudget"] = None
|
|
233
|
-
|
|
234
|
-
# ── step observer port (optional) ────────────────────────────────
|
|
235
|
-
# Default per-runtime observer for live step events; a per-run observer
|
|
236
|
-
# can also be attached on AgentRunContext.on_step. Both are advisory.
|
|
237
|
-
on_step: Optional[Callable[[Dict[str, Any]], None]] = None
|
|
238
|
-
|
|
239
|
-
# ── agent profile (optional, v9.9.7) ─────────────────────────────
|
|
240
|
-
# How hard the loop works to keep a weak model on contract. None selects
|
|
241
|
-
# per-run from the executing model id (``profile_for_model``); tests
|
|
242
|
-
# inject a fixed profile.
|
|
243
|
-
agent_profile: Optional["AgentProfile"] = None
|
|
244
|
-
|
|
245
|
-
# ── permission mode port (optional, v9.9.8) ──────────────────────
|
|
246
|
-
# The autonomy dial. Either a static mode, or a resolver callable that
|
|
247
|
-
# accepts ``user_email``/``workspace_id`` scope kwargs (preferred) or no
|
|
248
|
-
# arguments at all — see ``call_mode_source``. ``None`` means strict,
|
|
249
|
-
# which is exactly the pre-9.9.8 behaviour.
|
|
250
|
-
permission_mode: Any = None
|
|
251
|
-
|
|
252
|
-
# ── Self-Model port (optional, v11.2.0) ──────────────────────────
|
|
253
|
-
# What the Brain has learned about its owner, injected into the executor
|
|
254
|
-
# prompt. Either a fixed string or a resolver callable taking
|
|
255
|
-
# ``user_email``/``workspace_id`` scope kwargs (or none at all). 11.1.0
|
|
256
|
-
# built ``executor_prompt_for(self_model_summary=…)`` and then had nothing
|
|
257
|
-
# to pass it, because the runtime held its executor prompt as a fixed
|
|
258
|
-
# string; this is the port that was missing. ``None`` — and an empty
|
|
259
|
-
# summary — produce exactly the prompt bytes the loop produced before it
|
|
260
|
-
# existed.
|
|
261
|
-
self_model_summary: Any = None
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
class SingleAgentRuntime:
|
|
265
|
-
"""Drives the agent state machine over injected :class:`AgentDeps`."""
|
|
266
|
-
|
|
267
|
-
def __init__(self, deps: AgentDeps) -> None:
|
|
268
|
-
self.deps = deps
|
|
269
|
-
self._env_phase_budgets: Optional[PhaseBudgets] = None
|
|
270
|
-
self._env_transcript_budget: Optional[TranscriptBudget] = None
|
|
271
|
-
|
|
272
|
-
@property
|
|
273
|
-
def phase_budgets(self) -> PhaseBudgets:
|
|
274
|
-
# getattr twice: partially-constructed runtimes/deps (tests build them
|
|
275
|
-
# via __new__ or minimal fakes) still get working default budgets.
|
|
276
|
-
injected = getattr(self.deps, "phase_budgets", None)
|
|
277
|
-
if injected is not None:
|
|
278
|
-
return injected
|
|
279
|
-
cached = getattr(self, "_env_phase_budgets", None)
|
|
280
|
-
if cached is None:
|
|
281
|
-
cached = PhaseBudgets.from_env()
|
|
282
|
-
self._env_phase_budgets = cached
|
|
283
|
-
return cached
|
|
284
|
-
|
|
285
|
-
@property
|
|
286
|
-
def transcript_budget(self) -> TranscriptBudget:
|
|
287
|
-
injected = getattr(self.deps, "transcript_budget", None)
|
|
288
|
-
if injected is not None:
|
|
289
|
-
return injected
|
|
290
|
-
cached = getattr(self, "_env_transcript_budget", None)
|
|
291
|
-
if cached is None:
|
|
292
|
-
cached = TranscriptBudget.from_env()
|
|
293
|
-
self._env_transcript_budget = cached
|
|
294
|
-
return cached
|
|
295
|
-
|
|
296
|
-
# ── permission mode (v9.9.8) ─────────────────────────────────────
|
|
297
|
-
def resolve_permission_mode(
|
|
298
|
-
self,
|
|
299
|
-
ctx: Optional[AgentRunContext] = None,
|
|
300
|
-
*,
|
|
301
|
-
user_email: Optional[str] = None,
|
|
302
|
-
workspace_id: Optional[str] = None,
|
|
303
|
-
) -> PermissionMode:
|
|
304
|
-
"""Autonomy dial for this run.
|
|
305
|
-
|
|
306
|
-
A mode stamped on ``ctx`` wins, so the plan a user approved and every
|
|
307
|
-
tool step in the same run are judged by one dial even if the stored
|
|
308
|
-
preference changes mid-run. Otherwise the resolver on ``deps`` is
|
|
309
|
-
consulted with the caller's scope — resolving unscoped would collapse
|
|
310
|
-
every caller onto the process-wide default.
|
|
311
|
-
"""
|
|
312
|
-
return resolve_deps_mode(
|
|
313
|
-
self.deps, ctx, user_email=user_email, workspace_id=workspace_id,
|
|
314
|
-
)
|
|
315
|
-
|
|
316
|
-
def _governed_path_exists(self, name: str, path: str) -> bool:
|
|
317
|
-
"""Does this tool call's *real* target already exist?
|
|
318
|
-
|
|
319
|
-
The document creators sanitize ``filename`` into their own output
|
|
320
|
-
directory, so the raw argument is resolved through
|
|
321
|
-
:func:`document_output_target` first — checking it verbatim would
|
|
322
|
-
inspect a path nothing ever writes and the fail-closed overwrite guard
|
|
323
|
-
would never fire. Workspace-relative paths resolve under
|
|
324
|
-
``deps.agent_root``; absolute paths (home-sandbox writes) are honored
|
|
325
|
-
as-is. Never raises: governance must not be able to crash the loop, and
|
|
326
|
-
an unresolvable path degrades to "new file", which the remaining gates
|
|
327
|
-
still cover.
|
|
328
|
-
"""
|
|
329
|
-
try:
|
|
330
|
-
candidate = Path(document_output_target(name, path) or path)
|
|
331
|
-
if not candidate.is_absolute():
|
|
332
|
-
candidate = Path(self.deps.agent_root) / candidate
|
|
333
|
-
return candidate.exists()
|
|
334
|
-
except Exception: # noqa: BLE001 — classification is best-effort
|
|
335
|
-
return False
|
|
336
|
-
|
|
337
|
-
def _governed_tools(self) -> FrozenSet[str]:
|
|
338
|
-
governor = getattr(self.deps, "change_governor", None)
|
|
339
|
-
if governor is None:
|
|
340
|
-
return frozenset()
|
|
341
|
-
return frozenset(getattr(governor, "governed_tools", frozenset()))
|
|
342
|
-
|
|
343
|
-
def profile_for(self, model_id: Optional[str]) -> AgentProfile:
|
|
344
|
-
"""Loop profile for the model actually executing this run (v9.9.7).
|
|
345
|
-
|
|
346
|
-
An injected ``AgentDeps.agent_profile`` wins (tests, explicit config);
|
|
347
|
-
otherwise the profile is derived from the model id, so a small local
|
|
348
|
-
model gets the compact loop without any extra configuration.
|
|
349
|
-
"""
|
|
350
|
-
injected = getattr(self.deps, "agent_profile", None)
|
|
351
|
-
if injected is not None:
|
|
352
|
-
return injected
|
|
353
|
-
return profile_for_model(model_id)
|
|
354
|
-
|
|
355
|
-
def _emit_step(self, ctx: AgentRunContext, phase: str, event: str, **details: Any) -> None:
|
|
356
|
-
"""Fire the per-run / deps step observers (review Wave 1.1).
|
|
357
|
-
|
|
358
|
-
Observers power the live step timeline in the UI. They are pure
|
|
359
|
-
telemetry: any observer failure is logged and swallowed — the loop
|
|
360
|
-
itself must never notice.
|
|
361
|
-
"""
|
|
362
|
-
payload: Dict[str, Any] = {"phase": phase, "event": event}
|
|
363
|
-
for key, value in details.items():
|
|
364
|
-
if value is not None:
|
|
365
|
-
payload[key] = value
|
|
366
|
-
for observer in (getattr(ctx, "on_step", None), getattr(self.deps, "on_step", None)):
|
|
367
|
-
if observer is None:
|
|
368
|
-
continue
|
|
369
|
-
try:
|
|
370
|
-
observer(dict(payload))
|
|
371
|
-
except Exception as exc: # noqa: BLE001 — observers are advisory
|
|
372
|
-
logging.warning("agent step observer failed: %s", exc)
|
|
373
|
-
|
|
374
|
-
@staticmethod
|
|
375
|
-
def _project_block(ctx: AgentRunContext) -> str:
|
|
376
|
-
"""Project-session context for prompts, or "" for a standalone run.
|
|
377
|
-
|
|
378
|
-
Multi-turn project loop (v9.9.6): a later run must see the files the
|
|
379
|
-
project already produced and what is still open, instead of planning
|
|
380
|
-
from a blank workspace every time.
|
|
381
|
-
"""
|
|
382
|
-
summary = str(getattr(ctx, "project_context", "") or "").strip()
|
|
383
|
-
return f"\n\n[PROJECT SESSION]\n{summary}" if summary else ""
|
|
384
|
-
|
|
385
|
-
def boundary(self) -> Dict[str, Any]:
|
|
386
|
-
return runtime_boundary_contract(
|
|
387
|
-
name="SingleAgentRuntime",
|
|
388
|
-
runtime="single_agent",
|
|
389
|
-
entrypoint="latticeai.core.agent.SingleAgentRuntime",
|
|
390
|
-
surface="/agent",
|
|
391
|
-
owns="single-agent PLAN / EXECUTE / VERIFY state machine over injected ports",
|
|
392
|
-
compatibility_aliases=[],
|
|
393
|
-
)
|
|
394
|
-
|
|
395
|
-
def config(self) -> Dict[str, Any]:
|
|
396
|
-
return {
|
|
397
|
-
"boundary": self.boundary(),
|
|
398
|
-
"states": [state.value for state in AgentState],
|
|
399
|
-
"terminal_states": sorted(state.value for state in AGENT_TERMINAL_STATES),
|
|
400
|
-
"execution_mode": "injected_ports",
|
|
401
|
-
}
|
|
402
|
-
|
|
403
|
-
def contract(self, ctx: AgentRunContext, req: Any, *, run_id: Optional[str] = None) -> Dict[str, Any]:
|
|
404
|
-
"""Expose the shared agent-run contract for the single-agent loop."""
|
|
405
|
-
return single_agent_contract(ctx=ctx, goal=getattr(req, "message", ""), run_id=run_id)
|
|
406
|
-
|
|
407
|
-
# ── PLAN ─────────────────────────────────────────────────────────
|
|
408
|
-
async def plan(
|
|
409
|
-
self, ctx: AgentRunContext, req: Any, lang_hint: str, current_user: str,
|
|
410
|
-
model_id: Optional[str] = None,
|
|
411
|
-
) -> None:
|
|
412
|
-
"""PLAN: Planner role produces a structured plan JSON."""
|
|
413
|
-
d = self.deps
|
|
414
|
-
project_block = self._project_block(ctx)
|
|
415
|
-
context = (
|
|
416
|
-
f"{d.planner_prompt}\n\n"
|
|
417
|
-
f"[LANGUAGE HINT: {lang_hint}]\n"
|
|
418
|
-
f"Workspace root: {d.agent_root}{project_block}\n\n"
|
|
419
|
-
f"User request: {req.message}"
|
|
420
|
-
)
|
|
421
|
-
raw = await d.generate_as(
|
|
422
|
-
model_id,
|
|
423
|
-
message="Produce a JSON execution plan for this request.",
|
|
424
|
-
context=context, max_tokens=self.phase_budgets.plan_tokens, temperature=0.1,
|
|
425
|
-
)
|
|
426
|
-
ctx.trace.llm_call("plan", model=model_id)
|
|
427
|
-
try:
|
|
428
|
-
plan, plan_repairs = extract_action_details(str(raw))
|
|
429
|
-
ctx.trace.repair("plan", repairs=plan_repairs)
|
|
430
|
-
except ValueError as exc:
|
|
431
|
-
ctx.trace.parse_error("plan", error=str(exc), recovered=True)
|
|
432
|
-
plan = {
|
|
433
|
-
"action": "plan", "state": "PLAN",
|
|
434
|
-
"goal": req.message, "steps": [],
|
|
435
|
-
"requires_approval": False, "rollback_strategy": "none", "estimated_steps": 1,
|
|
436
|
-
}
|
|
437
|
-
plan, plan_fixes = normalize_plan(plan, req.message)
|
|
438
|
-
if plan_fixes:
|
|
439
|
-
ctx.trace.repair("plan", repairs=plan_fixes)
|
|
440
|
-
ctx.plan = plan
|
|
441
|
-
ctx.transcript.append({
|
|
442
|
-
"state": AgentState.PLANNING.value,
|
|
443
|
-
"goal": plan.get("goal", req.message),
|
|
444
|
-
"steps": plan.get("steps", []),
|
|
445
|
-
"requires_approval": plan.get("requires_approval", False),
|
|
446
|
-
"rollback_strategy": plan.get("rollback_strategy", "none"),
|
|
447
|
-
"estimated_steps": plan.get("estimated_steps", 1),
|
|
448
|
-
**({"plan_fixes": plan_fixes} if plan_fixes else {}),
|
|
449
|
-
})
|
|
450
|
-
self._emit_step(
|
|
451
|
-
ctx, "plan", "planned",
|
|
452
|
-
goal=str(plan.get("goal") or "")[:200],
|
|
453
|
-
steps=len(plan.get("steps") or []),
|
|
454
|
-
requires_approval=bool(plan.get("requires_approval", False)),
|
|
455
|
-
)
|
|
456
|
-
ctx.state = AgentState.WAITING_APPROVAL
|
|
457
|
-
|
|
458
|
-
# ── APPROVAL ─────────────────────────────────────────────────────
|
|
459
|
-
def approval_requirements(self, ctx: AgentRunContext) -> Dict[str, Any]:
|
|
460
|
-
"""Read-only preview of the approval gate for a planned run.
|
|
461
|
-
|
|
462
|
-
Shares the exact predicate :meth:`approve` enforces, so the HTTP
|
|
463
|
-
layer can pause a run as ``awaiting_approval`` (with a plan summary
|
|
464
|
-
for the user) instead of letting it fail closed — without ever
|
|
465
|
-
weakening the gate itself.
|
|
466
|
-
"""
|
|
467
|
-
d = self.deps
|
|
468
|
-
mode = self.resolve_permission_mode(ctx)
|
|
469
|
-
# Governor-managed tools never hard-block the plan: each call is
|
|
470
|
-
# classified at execution time — additive creates run, mutations and
|
|
471
|
-
# deletions of existing content become review proposals.
|
|
472
|
-
governed_tools = self._governed_tools()
|
|
473
|
-
steps = ctx.plan.get("steps", [])
|
|
474
|
-
non_auto = non_auto_plan_steps(
|
|
475
|
-
mode, steps, d.tool_governance or {}, governed_tools=governed_tools,
|
|
476
|
-
)
|
|
477
|
-
requires = plan_requires_approval(
|
|
478
|
-
mode,
|
|
479
|
-
non_auto_steps=non_auto,
|
|
480
|
-
plan_flag=bool(ctx.plan.get("requires_approval", False)),
|
|
481
|
-
)
|
|
482
|
-
lines = [
|
|
483
|
-
f"{index}. {step.get('description') or step.get('action') or '?'}"
|
|
484
|
-
for index, step in enumerate(steps, start=1)
|
|
485
|
-
]
|
|
486
|
-
summary = str(ctx.plan.get("goal") or "").strip()
|
|
487
|
-
if lines:
|
|
488
|
-
summary = (summary + "\n" if summary else "") + "\n".join(lines)
|
|
489
|
-
return {
|
|
490
|
-
"requires_approval": requires,
|
|
491
|
-
"non_auto_steps": non_auto,
|
|
492
|
-
"permission_mode": mode.value,
|
|
493
|
-
"plan_summary": summary,
|
|
494
|
-
}
|
|
495
|
-
|
|
496
|
-
def approve(self, ctx: AgentRunContext, current_user: str, *, approved_by_human: bool = False) -> None:
|
|
497
|
-
"""APPROVAL: Check governance, log decision, auto-approve (future: UI prompt)."""
|
|
498
|
-
d = self.deps
|
|
499
|
-
requirements = self.approval_requirements(ctx)
|
|
500
|
-
non_auto = requirements["non_auto_steps"]
|
|
501
|
-
requires = requirements["requires_approval"]
|
|
502
|
-
|
|
503
|
-
ctx.transcript.append({
|
|
504
|
-
"state": AgentState.WAITING_APPROVAL.value,
|
|
505
|
-
"requires_approval": requires,
|
|
506
|
-
"non_auto_approve_steps": non_auto,
|
|
507
|
-
"decision": "human_approved" if requires and approved_by_human else ("blocked_pending_approval" if requires else "auto_approved"),
|
|
508
|
-
})
|
|
509
|
-
decision = "human_approved" if requires and approved_by_human else ("blocked_pending_approval" if requires else "auto_approved")
|
|
510
|
-
ctx.trace.decision("approve", decision=decision, non_auto_steps=len(non_auto))
|
|
511
|
-
self._emit_step(ctx, "approval", "decision", decision=decision)
|
|
512
|
-
d.audit(
|
|
513
|
-
"agent_approval", user_email=current_user,
|
|
514
|
-
requires_approval=requires,
|
|
515
|
-
non_auto_steps=non_auto,
|
|
516
|
-
decision=decision,
|
|
517
|
-
)
|
|
518
|
-
if requires and not approved_by_human:
|
|
519
|
-
ctx.final_message = (
|
|
520
|
-
"이 작업에는 명시 승인이 필요한 도구가 포함되어 있어 자동 실행을 중단했습니다. "
|
|
521
|
-
"human_in_loop 승인 흐름으로 다시 실행해 주세요."
|
|
522
|
-
)
|
|
523
|
-
ctx.state = AgentState.FAILED
|
|
524
|
-
return
|
|
525
|
-
ctx.approved_by_human = bool(approved_by_human)
|
|
526
|
-
ctx.state = AgentState.EXECUTING
|
|
527
|
-
|
|
528
|
-
# ── EXECUTE ──────────────────────────────────────────────────────
|
|
529
|
-
async def execute(
|
|
530
|
-
self, ctx: AgentRunContext, req: Any, lang_hint: str,
|
|
531
|
-
current_user: str, max_steps: int, model_id: Optional[str] = None,
|
|
532
|
-
) -> None:
|
|
533
|
-
"""EXECUTE: Executor role calls tools one at a time until final or budget exhausted."""
|
|
534
|
-
d = self.deps
|
|
535
|
-
profile = self.profile_for(model_id)
|
|
536
|
-
exec_count = sum(1 for s in ctx.transcript if s.get("state") == AgentState.EXECUTING.value)
|
|
537
|
-
budget = max(1, max_steps - exec_count)
|
|
538
|
-
parse_failures = 0
|
|
539
|
-
|
|
540
|
-
for _ in range(budget):
|
|
541
|
-
request_workspace = getattr(req, "workspace_id", None)
|
|
542
|
-
context = self._executor_context(
|
|
543
|
-
ctx, req, lang_hint, current_user, request_workspace, profile=profile
|
|
544
|
-
)
|
|
545
|
-
raw = await d.generate_as(
|
|
546
|
-
model_id,
|
|
547
|
-
message="Execute the next step.",
|
|
548
|
-
context=context, max_tokens=self.phase_budgets.execute_tokens,
|
|
549
|
-
temperature=req.temperature,
|
|
550
|
-
)
|
|
551
|
-
ctx.trace.llm_call("execute", model=model_id)
|
|
552
|
-
try:
|
|
553
|
-
action, exec_repairs = extract_action_details(str(raw))
|
|
554
|
-
ctx.trace.repair("execute", repairs=exec_repairs)
|
|
555
|
-
except ValueError as exc:
|
|
556
|
-
parse_failures += 1
|
|
557
|
-
if self._note_parse_failure(ctx, raw, exc, parse_failures, profile):
|
|
558
|
-
# Direct-path fallback (v9.9.7): a small model that cannot
|
|
559
|
-
# hold the tool-call protocol can still write a file. Run
|
|
560
|
-
# the plan's own file steps without asking for any JSON.
|
|
561
|
-
if profile.direct_path_fallback and await self._direct_file_path(
|
|
562
|
-
ctx, req, current_user, model_id
|
|
563
|
-
):
|
|
564
|
-
ctx.state = AgentState.VERIFYING
|
|
565
|
-
return
|
|
566
|
-
break
|
|
567
|
-
continue
|
|
568
|
-
|
|
569
|
-
name = str(action.get("action") or "")
|
|
570
|
-
thoughts = str(action.get("thoughts") or "")[:600]
|
|
571
|
-
args = action.get("args") or {}
|
|
572
|
-
|
|
573
|
-
if name in SCOPED_KNOWLEDGE_TOOLS:
|
|
574
|
-
# Scope is server-owned, never model-owned. Overwrite any
|
|
575
|
-
# claimed values before policy evaluation, audit, and dispatch.
|
|
576
|
-
args = dict(args)
|
|
577
|
-
args["workspace_id"] = request_workspace or "personal"
|
|
578
|
-
args["user_email"] = current_user or "local"
|
|
579
|
-
|
|
580
|
-
if name == "final":
|
|
581
|
-
ctx.final_message = action.get("message", "작업을 완료했습니다.")
|
|
582
|
-
ctx.transcript.append({
|
|
583
|
-
"state": AgentState.EXECUTING.value, "action": "final", "thoughts": thoughts,
|
|
584
|
-
})
|
|
585
|
-
ctx.trace.decision("execute", decision="final")
|
|
586
|
-
self._emit_step(ctx, "execute", "final")
|
|
587
|
-
ctx.state = AgentState.VERIFYING
|
|
588
|
-
return
|
|
589
|
-
|
|
590
|
-
# Loop guard
|
|
591
|
-
if self._is_repeated_create(ctx, name, args):
|
|
592
|
-
ctx.transcript.append({
|
|
593
|
-
"state": AgentState.EXECUTING.value, "action": name,
|
|
594
|
-
"error": "LOOP_DETECTED: identical action+args repeated — halted.",
|
|
595
|
-
})
|
|
596
|
-
ctx.trace.decision("execute", decision="loop_detected", tool=name)
|
|
597
|
-
self._emit_step(ctx, "execute", "blocked", action=name, reason="loop_detected")
|
|
598
|
-
break
|
|
599
|
-
|
|
600
|
-
if name == "clear_history":
|
|
601
|
-
result = d.clear_history(args.get("keep_last", 0))
|
|
602
|
-
ctx.transcript.append({
|
|
603
|
-
"state": AgentState.EXECUTING.value, "action": name,
|
|
604
|
-
"thoughts": thoughts, "args": args, "result": result,
|
|
605
|
-
})
|
|
606
|
-
self._emit_step(ctx, "execute", "tool", action=name, ok=True)
|
|
607
|
-
continue
|
|
608
|
-
|
|
609
|
-
policy = d.policy_for(name, args)
|
|
610
|
-
risk = d.risk_level(policy)
|
|
611
|
-
|
|
612
|
-
proposed, governor_allows_additive = self._governor_review(
|
|
613
|
-
ctx, name, thoughts, args, policy, risk, current_user, request_workspace,
|
|
614
|
-
conversation_id=getattr(req, "conversation_id", None),
|
|
615
|
-
)
|
|
616
|
-
if proposed:
|
|
617
|
-
continue
|
|
618
|
-
|
|
619
|
-
if self._blocked_by_gates(
|
|
620
|
-
ctx, req, name, thoughts, args, policy, risk,
|
|
621
|
-
current_user, governor_allows_additive,
|
|
622
|
-
):
|
|
623
|
-
continue
|
|
624
|
-
|
|
625
|
-
self._dispatch_step(ctx, name, thoughts, args, policy, risk, current_user)
|
|
626
|
-
|
|
627
|
-
ctx.state = AgentState.VERIFYING
|
|
628
|
-
|
|
629
|
-
def _self_model_summary(
|
|
630
|
-
self, ctx: AgentRunContext, current_user: str, request_workspace: Optional[str]
|
|
631
|
-
) -> str:
|
|
632
|
-
"""What this Brain knows about its owner, resolved once per run.
|
|
633
|
-
|
|
634
|
-
The port may be a plain string or a resolver taking scope kwargs — the
|
|
635
|
-
same two shapes ``permission_mode`` accepts, so there is one convention
|
|
636
|
-
for "a value or a way to get one". Anything that goes wrong yields an
|
|
637
|
-
empty summary: prompt assembly must not fail because a profile could
|
|
638
|
-
not be read, and an empty summary is byte-identical to having no port
|
|
639
|
-
at all.
|
|
640
|
-
"""
|
|
641
|
-
if ctx.self_model_summary is not None:
|
|
642
|
-
return ctx.self_model_summary
|
|
643
|
-
source = self.deps.self_model_summary
|
|
644
|
-
summary = ""
|
|
645
|
-
if source is not None:
|
|
646
|
-
try:
|
|
647
|
-
if callable(source):
|
|
648
|
-
try:
|
|
649
|
-
summary = source(
|
|
650
|
-
user_email=current_user or None,
|
|
651
|
-
workspace_id=request_workspace,
|
|
652
|
-
)
|
|
653
|
-
except TypeError:
|
|
654
|
-
# A resolver that takes no scope arguments is allowed.
|
|
655
|
-
summary = source()
|
|
656
|
-
else:
|
|
657
|
-
summary = source
|
|
658
|
-
except Exception: # noqa: BLE001 — a profile is never worth a failed run
|
|
659
|
-
logging.debug("agent: self-model summary unavailable", exc_info=True)
|
|
660
|
-
summary = ""
|
|
661
|
-
ctx.self_model_summary = str(summary or "").strip()
|
|
662
|
-
return ctx.self_model_summary
|
|
663
|
-
|
|
664
|
-
def _executor_context(
|
|
665
|
-
self, ctx: AgentRunContext, req: Any, lang_hint: str,
|
|
666
|
-
current_user: str, request_workspace: Optional[str],
|
|
667
|
-
profile: Optional[AgentProfile] = None,
|
|
668
|
-
) -> str:
|
|
669
|
-
"""Assemble one executor turn's prompt (plan, corrections, recent chat)."""
|
|
670
|
-
d = self.deps
|
|
671
|
-
# Only the latest corrections steer the next attempt — stale hints
|
|
672
|
-
# from earlier retries dilute weak models (review Wave 0.3).
|
|
673
|
-
active_corrections = ctx.corrections[-3:]
|
|
674
|
-
corrections_hint = (
|
|
675
|
-
"\n\nCritic corrections from previous attempt:\n"
|
|
676
|
-
+ "\n".join(f"- {c}" for c in active_corrections)
|
|
677
|
-
) if active_corrections else ""
|
|
678
|
-
|
|
679
|
-
recent_kwargs = {
|
|
680
|
-
"conversation_id": req.conversation_id,
|
|
681
|
-
"user_email": current_user or None,
|
|
682
|
-
}
|
|
683
|
-
if request_workspace is not None:
|
|
684
|
-
recent_kwargs["workspace_id"] = request_workspace
|
|
685
|
-
recent_conversation = d.recent_chat_context(**recent_kwargs) or "(none)"
|
|
686
|
-
budget = self.transcript_budget
|
|
687
|
-
# A small model drowns in a long transcript far sooner than a large
|
|
688
|
-
# one, so the profile may narrow the window (v9.9.7).
|
|
689
|
-
window = min(budget.window, profile.transcript_window) if profile else budget.window
|
|
690
|
-
bounded_transcript = compact_transcript(
|
|
691
|
-
ctx.transcript,
|
|
692
|
-
window=window,
|
|
693
|
-
result_chars=budget.result_chars,
|
|
694
|
-
)
|
|
695
|
-
# Mid-run workspace awareness (review L5): later steps must see what
|
|
696
|
-
# this run already produced instead of a stale workspace picture.
|
|
697
|
-
written = files_written(ctx.transcript, d.file_create_actions)
|
|
698
|
-
written_hint = (
|
|
699
|
-
"\n\nFiles written by this run so far (they exist in the workspace now):\n"
|
|
700
|
-
+ "\n".join(f"- {path}" for path in written)
|
|
701
|
-
) if written else ""
|
|
702
|
-
return (
|
|
703
|
-
# v11.1.0: the executor prompt carries profile-aware file-writing
|
|
704
|
-
# hints, because "wrote nothing at all" was the weak-model failure
|
|
705
|
-
# mode the loop could not repair after the fact.
|
|
706
|
-
f"{executor_prompt_for(d.executor_prompt, profile=profile, self_model_summary=self._self_model_summary(ctx, current_user, request_workspace))}\n\n"
|
|
707
|
-
f"[LANGUAGE HINT: {lang_hint}]\n"
|
|
708
|
-
f"Workspace root: {d.agent_root}{self._project_block(ctx)}\n\n"
|
|
709
|
-
f"PLAN:\n{json.dumps(ctx.plan, ensure_ascii=False)}{written_hint}\n\n"
|
|
710
|
-
f"Recent conversation:\n{recent_conversation}\n\n"
|
|
711
|
-
f"User request: {req.message}{corrections_hint}\n\n"
|
|
712
|
-
f"Execution transcript:\n{json.dumps(bounded_transcript, ensure_ascii=False, indent=2)}"
|
|
713
|
-
)
|
|
714
|
-
|
|
715
|
-
def _note_parse_failure(
|
|
716
|
-
self, ctx: AgentRunContext, raw: Any, exc: ValueError, parse_failures: int,
|
|
717
|
-
profile: Optional[AgentProfile] = None,
|
|
718
|
-
) -> bool:
|
|
719
|
-
"""Record one executor parse slip; True when the run should stop retrying."""
|
|
720
|
-
profile = profile or self.profile_for(None)
|
|
721
|
-
ctx.transcript.append({
|
|
722
|
-
"state": AgentState.EXECUTING.value, "action": "parse_error",
|
|
723
|
-
"raw": str(raw)[:400], "error": str(exc),
|
|
724
|
-
})
|
|
725
|
-
if parse_failures >= profile.parse_failure_budget:
|
|
726
|
-
ctx.trace.parse_error("execute", error=str(exc), recovered=False)
|
|
727
|
-
self._emit_step(ctx, "execute", "parse_error", recovered=False)
|
|
728
|
-
return True
|
|
729
|
-
ctx.trace.parse_error("execute", error=str(exc), recovered=True)
|
|
730
|
-
self._emit_step(ctx, "execute", "parse_error", recovered=True)
|
|
731
|
-
# Weak models often need one concrete reminder of the wire
|
|
732
|
-
# format; feed it through the corrections channel and retry
|
|
733
|
-
# instead of aborting the whole run on the first slip.
|
|
734
|
-
hint = (
|
|
735
|
-
'Your last reply was not a single JSON action object. Reply with '
|
|
736
|
-
'EXACTLY one JSON object like {"thoughts": "...", "action": '
|
|
737
|
-
'"tool_name", "args": {...}} and nothing else.'
|
|
738
|
-
)
|
|
739
|
-
if parse_failures >= profile.escalate_after:
|
|
740
|
-
# Escalate: name the valid tools so the model stops
|
|
741
|
-
# inventing action names or prose. The compact profile escalates
|
|
742
|
-
# a slip earlier — a small model needs the list sooner.
|
|
743
|
-
valid = ", ".join(sorted(self.deps.tool_governance.keys()))
|
|
744
|
-
hint = (
|
|
745
|
-
f"{hint} Valid action values are: {valid}, final. "
|
|
746
|
-
'Use {"action": "final", "message": "..."} to finish.'
|
|
747
|
-
)
|
|
748
|
-
if hint not in ctx.corrections:
|
|
749
|
-
ctx.corrections.append(hint)
|
|
750
|
-
ctx.trace.correction("execute", hint=hint)
|
|
751
|
-
return False
|
|
752
|
-
|
|
753
|
-
async def _direct_file_path(
|
|
754
|
-
self, ctx: AgentRunContext, req: Any, current_user: str,
|
|
755
|
-
model_id: Optional[str],
|
|
756
|
-
) -> bool:
|
|
757
|
-
"""Write the plan's file steps without asking the model for JSON (v9.9.7).
|
|
758
|
-
|
|
759
|
-
The compact profile's escape hatch. A 1–4B local model that cannot hold
|
|
760
|
-
the tool-call protocol can still write a file, so when JSON tool calls
|
|
761
|
-
are exhausted the loop drops the protocol entirely: it takes the paths
|
|
762
|
-
the *planner* already chose and asks only for file content in plain
|
|
763
|
-
text, through the same validated
|
|
764
|
-
:func:`~latticeai.core.file_generation.generate_file_content` pipeline
|
|
765
|
-
the direct chat path uses.
|
|
766
|
-
|
|
767
|
-
Returns True when at least one file was actually written. Honest
|
|
768
|
-
failure modes: no planned paths, a governor that stages the write as a
|
|
769
|
-
proposal, or a tool error all return False and leave the run to end as
|
|
770
|
-
it would have — this never fabricates evidence.
|
|
771
|
-
"""
|
|
772
|
-
d = self.deps
|
|
773
|
-
planned: List[str] = []
|
|
774
|
-
for step in ctx.plan.get("steps") or []:
|
|
775
|
-
if not isinstance(step, dict) or step.get("action") not in d.file_create_actions:
|
|
776
|
-
continue
|
|
777
|
-
path = str((step.get("args") or {}).get("path") or "").strip()
|
|
778
|
-
if path and path not in planned:
|
|
779
|
-
planned.append(path)
|
|
780
|
-
if not planned:
|
|
781
|
-
inferred = infer_file_target(getattr(req, "message", "") or "")
|
|
782
|
-
if inferred:
|
|
783
|
-
planned = [inferred]
|
|
784
|
-
if not planned:
|
|
785
|
-
return False
|
|
786
|
-
|
|
787
|
-
goal = str(ctx.plan.get("goal") or getattr(req, "message", "") or "")
|
|
788
|
-
|
|
789
|
-
async def _generate(context: str) -> Any:
|
|
790
|
-
return await d.generate_as(
|
|
791
|
-
model_id,
|
|
792
|
-
message="Write the file content.",
|
|
793
|
-
context=context,
|
|
794
|
-
max_tokens=self.phase_budgets.execute_tokens,
|
|
795
|
-
temperature=0.2,
|
|
796
|
-
)
|
|
797
|
-
|
|
798
|
-
wrote = False
|
|
799
|
-
for path in planned[:6]:
|
|
800
|
-
try:
|
|
801
|
-
content, meta = await generate_file_content(
|
|
802
|
-
_generate,
|
|
803
|
-
target_path=path,
|
|
804
|
-
user_request=goal,
|
|
805
|
-
bundle_files=planned if len(planned) > 1 else None,
|
|
806
|
-
)
|
|
807
|
-
except Exception as exc: # noqa: BLE001 — fallback must not raise
|
|
808
|
-
logging.warning("direct file path generation failed for %s: %s", path, exc)
|
|
809
|
-
continue
|
|
810
|
-
ctx.trace.llm_call("execute", model=model_id)
|
|
811
|
-
ctx.trace.repair("execute", repairs=["direct_path_fallback"])
|
|
812
|
-
args = {"path": path, "content": content}
|
|
813
|
-
policy = d.policy_for("write_file", args)
|
|
814
|
-
risk = d.risk_level(policy)
|
|
815
|
-
before = len(ctx.transcript)
|
|
816
|
-
self._dispatch_step(ctx, "write_file", "direct path fallback", args, policy, risk, current_user)
|
|
817
|
-
last = ctx.transcript[-1] if len(ctx.transcript) > before else {}
|
|
818
|
-
if isinstance(last.get("result"), dict) and not last["result"].get("proposed"):
|
|
819
|
-
wrote = True
|
|
820
|
-
last["direct_path"] = True
|
|
821
|
-
last["generation"] = {"repaired": bool(meta.get("repaired"))}
|
|
822
|
-
if wrote:
|
|
823
|
-
ctx.trace.decision("execute", decision="direct_path_fallback", files=len(planned))
|
|
824
|
-
self._emit_step(ctx, "execute", "direct_path", files=len(planned))
|
|
825
|
-
ctx.final_message = (
|
|
826
|
-
"도구 호출 형식을 계속 벗어나서, 계획에 있던 파일을 직접 생성했습니다. "
|
|
827
|
-
"내용을 확인해 주세요."
|
|
828
|
-
)
|
|
829
|
-
return wrote
|
|
830
|
-
|
|
831
|
-
def _is_repeated_create(self, ctx: AgentRunContext, name: Any, args: dict) -> bool:
|
|
832
|
-
"""Loop guard: the same file-create action+args re-issued right after a result."""
|
|
833
|
-
exec_steps = [s for s in ctx.transcript if s.get("state") == AgentState.EXECUTING.value]
|
|
834
|
-
last = exec_steps[-1] if exec_steps else None
|
|
835
|
-
return bool(
|
|
836
|
-
name in self.deps.file_create_actions and last
|
|
837
|
-
and last.get("action") == name
|
|
838
|
-
and (last.get("args") or {}) == args
|
|
839
|
-
and "result" in last
|
|
840
|
-
)
|
|
841
|
-
|
|
842
|
-
def _governor_review(
|
|
843
|
-
self, ctx: AgentRunContext, name: str, thoughts: str, args: dict,
|
|
844
|
-
policy: Mapping[str, Any], risk: str, current_user: str, request_workspace: Optional[str],
|
|
845
|
-
conversation_id: Optional[str] = None,
|
|
846
|
-
) -> Tuple[bool, bool]:
|
|
847
|
-
"""Central change-class governance: create-new runs with minimal
|
|
848
|
-
friction, change/delete-existing becomes a review proposal.
|
|
849
|
-
|
|
850
|
-
Returns ``(proposed, governor_allows_additive)``: ``proposed`` means the
|
|
851
|
-
step was staged as a proposal (skip execution); ``allows_additive`` lets
|
|
852
|
-
an additive create pass the classic approval gate.
|
|
853
|
-
|
|
854
|
-
Under a mode that does not stage proposals (``trusted`` / ``bypass``)
|
|
855
|
-
the decision is made *before* the governor is consulted, because
|
|
856
|
-
``review`` persists a proposal as a side effect — reviewing first and
|
|
857
|
-
discarding the verdict afterwards would apply the change *and* leave an
|
|
858
|
-
orphan proposal pending in the Review Center.
|
|
859
|
-
"""
|
|
860
|
-
d = self.deps
|
|
861
|
-
if d.change_governor is None:
|
|
862
|
-
return False, False
|
|
863
|
-
|
|
864
|
-
mode = self.resolve_permission_mode(
|
|
865
|
-
ctx, user_email=current_user, workspace_id=request_workspace,
|
|
866
|
-
)
|
|
867
|
-
if not should_stage_proposal(mode, proposal_required=True):
|
|
868
|
-
if name not in self._governed_tools():
|
|
869
|
-
return False, False
|
|
870
|
-
if policy.get("destructive") or policy.get("risk") == "destructive":
|
|
871
|
-
# Let the destructive gate downstream own the block + transcript.
|
|
872
|
-
return False, False
|
|
873
|
-
d.audit(
|
|
874
|
-
"agent_change_auto_applied",
|
|
875
|
-
user_email=current_user,
|
|
876
|
-
workspace_id=request_workspace,
|
|
877
|
-
action=name,
|
|
878
|
-
path=str(args.get("path") or "") or None,
|
|
879
|
-
permission_mode=mode.value,
|
|
880
|
-
note="permission mode auto-applies mutation with audit",
|
|
881
|
-
)
|
|
882
|
-
return False, True
|
|
883
|
-
|
|
884
|
-
verdict = d.change_governor.review(
|
|
885
|
-
name, args, policy=dict(policy),
|
|
886
|
-
user_email=current_user, workspace_id=request_workspace,
|
|
887
|
-
conversation_id=conversation_id,
|
|
888
|
-
)
|
|
889
|
-
if verdict is not None and verdict.get("decision") == "proposed":
|
|
890
|
-
proposal = verdict.get("proposal") or {}
|
|
891
|
-
ctx.trace.tool("execute", name=name, outcome="proposed", risk=risk)
|
|
892
|
-
self._emit_step(ctx, "execute", "proposed", action=name)
|
|
893
|
-
ctx.transcript.append({
|
|
894
|
-
"state": AgentState.EXECUTING.value, "action": name,
|
|
895
|
-
"thoughts": thoughts, "args": {k: v for k, v in args.items() if k != "content"},
|
|
896
|
-
"risk": risk, "governance": dict(policy),
|
|
897
|
-
"result": {
|
|
898
|
-
"proposed": True,
|
|
899
|
-
"proposal_id": proposal.get("id"),
|
|
900
|
-
"note": "기존 내용을 바꾸는 작업이라 변경 제안으로 저장했습니다. 검토함에서 승인하면 적용됩니다.",
|
|
901
|
-
},
|
|
902
|
-
})
|
|
903
|
-
d.audit(
|
|
904
|
-
"agent_change_proposed", user_email=current_user,
|
|
905
|
-
action=name, proposal_id=proposal.get("id"),
|
|
906
|
-
change_class=(verdict.get("classification") or {}).get("change_class"),
|
|
907
|
-
)
|
|
908
|
-
return True, False
|
|
909
|
-
return False, (verdict is not None and verdict.get("decision") == "allow_additive")
|
|
910
|
-
|
|
911
|
-
def _blocked_by_gates(
|
|
912
|
-
self, ctx: AgentRunContext, req: Any, name: str, thoughts: str, args: dict,
|
|
913
|
-
policy: Mapping[str, Any], risk: str, current_user: str, governor_allows_additive: bool,
|
|
914
|
-
) -> bool:
|
|
915
|
-
"""Destructive / circuit-breaker / fail-closed-overwrite / approval gates.
|
|
916
|
-
|
|
917
|
-
Returns True when the step was blocked. The active permission mode can
|
|
918
|
-
widen what runs without an extra approval prompt, but never widens a
|
|
919
|
-
circuit breaker, the destructive gate, or the overwrite check.
|
|
920
|
-
"""
|
|
921
|
-
d = self.deps
|
|
922
|
-
mode = self.resolve_permission_mode(
|
|
923
|
-
ctx,
|
|
924
|
-
user_email=current_user,
|
|
925
|
-
workspace_id=getattr(req, "workspace_id", None),
|
|
926
|
-
)
|
|
927
|
-
# Hard denials first — mode-invariant. A circuit breaker (root/home
|
|
928
|
-
# paths, `rm -rf /` style commands) and a destructive policy are both
|
|
929
|
-
# audited as ``blocked`` with the reason that actually fired, rather
|
|
930
|
-
# than being flattened into the approval path.
|
|
931
|
-
breaker = is_circuit_breaker(name, policy, args)
|
|
932
|
-
hard_deny = breaker or (
|
|
933
|
-
"destructive policy"
|
|
934
|
-
if policy["risk"] == "destructive" or policy.get("destructive")
|
|
935
|
-
else None
|
|
936
|
-
)
|
|
937
|
-
if hard_deny:
|
|
938
|
-
error = (
|
|
939
|
-
f"BLOCKED: destructive action '{name}' not permitted in agent mode."
|
|
940
|
-
if hard_deny == "destructive policy"
|
|
941
|
-
else f"BLOCKED: {hard_deny}"
|
|
942
|
-
)
|
|
943
|
-
ctx.trace.tool("execute", name=name, outcome="blocked_destructive", risk=risk)
|
|
944
|
-
self._emit_step(ctx, "execute", "blocked", action=name, reason="destructive")
|
|
945
|
-
ctx.transcript.append({
|
|
946
|
-
"state": AgentState.EXECUTING.value, "action": name,
|
|
947
|
-
"thoughts": thoughts, "args": args, "risk": risk,
|
|
948
|
-
"governance": dict(policy),
|
|
949
|
-
"permission_mode": mode.value,
|
|
950
|
-
"error": error,
|
|
951
|
-
})
|
|
952
|
-
d.audit(
|
|
953
|
-
"agent_blocked", user_email=current_user, source=getattr(req, "source", None) or "agent",
|
|
954
|
-
action=name, reason="destructive", governance=dict(policy),
|
|
955
|
-
)
|
|
956
|
-
return True
|
|
957
|
-
|
|
958
|
-
# Fail-closed overwrite guard — mode-invariant, like the two above.
|
|
959
|
-
# A call that rewrites existing content but cannot be staged as a
|
|
960
|
-
# reviewable proposal (binary document creators, home-sandbox writes)
|
|
961
|
-
# has no safe apply path in ANY mode: trusted/bypass skip the approval
|
|
962
|
-
# *prompt*, they never remove the existence check. Without this the
|
|
963
|
-
# loop silently overwrote files that the HTTP surface refuses with 409
|
|
964
|
-
# (``ToolDispatchService.enforce_policy``).
|
|
965
|
-
overwrite = classify_tool_call(
|
|
966
|
-
name, args, policy=dict(policy),
|
|
967
|
-
path_exists=lambda candidate: self._governed_path_exists(name, candidate),
|
|
968
|
-
)
|
|
969
|
-
if overwrite.get("fail_closed"):
|
|
970
|
-
target = str(args.get("path") or args.get("filename") or "")
|
|
971
|
-
error = (
|
|
972
|
-
f"NEEDS_REVIEW: '{name}' 은(는) 이미 있는 파일 '{target}' 을(를) 덮어씁니다. "
|
|
973
|
-
"이 도구의 변경은 검토 가능한 제안으로 만들 수 없어 실행하지 않았습니다. "
|
|
974
|
-
"새 파일 이름으로 만들거나 write_file/edit_file 로 수정하세요."
|
|
975
|
-
)
|
|
976
|
-
ctx.trace.tool("execute", name=name, outcome="blocked_overwrite", risk=risk)
|
|
977
|
-
self._emit_step(ctx, "execute", "blocked", action=name, reason="overwrite")
|
|
978
|
-
ctx.transcript.append({
|
|
979
|
-
"state": AgentState.EXECUTING.value, "action": name,
|
|
980
|
-
"thoughts": thoughts,
|
|
981
|
-
# Same shape as a staged proposal: the payload is never worth
|
|
982
|
-
# replaying into the transcript, only the decision is.
|
|
983
|
-
"args": {k: v for k, v in args.items() if k != "content"},
|
|
984
|
-
"risk": risk,
|
|
985
|
-
"governance": dict(policy),
|
|
986
|
-
"permission_mode": mode.value,
|
|
987
|
-
"change_class": overwrite.get("change_class"),
|
|
988
|
-
"error": error,
|
|
989
|
-
})
|
|
990
|
-
d.audit(
|
|
991
|
-
"agent_blocked", user_email=current_user,
|
|
992
|
-
source=getattr(req, "source", None) or "agent",
|
|
993
|
-
action=name, reason="overwrite_fail_closed",
|
|
994
|
-
path=target or None,
|
|
995
|
-
change_class=overwrite.get("change_class"),
|
|
996
|
-
permission_mode=mode.value,
|
|
997
|
-
governance=dict(policy),
|
|
998
|
-
)
|
|
999
|
-
return True
|
|
1000
|
-
|
|
1001
|
-
reason = block_reason_for_tool(
|
|
1002
|
-
mode, name, policy, args,
|
|
1003
|
-
approved_by_human=bool(ctx.approved_by_human),
|
|
1004
|
-
governor_allows_additive=governor_allows_additive,
|
|
1005
|
-
)
|
|
1006
|
-
if reason is None:
|
|
1007
|
-
return False
|
|
1008
|
-
|
|
1009
|
-
d.audit(
|
|
1010
|
-
"agent_exec", user_email=current_user, source=getattr(req, "source", None) or "agent",
|
|
1011
|
-
state=AgentState.EXECUTING.value, action=name, risk=risk,
|
|
1012
|
-
shell=policy["shell"], network=policy["network"],
|
|
1013
|
-
destructive=policy["destructive"], sandbox=policy["sandbox"],
|
|
1014
|
-
rollback=policy["rollback"],
|
|
1015
|
-
permission_mode=mode.value,
|
|
1016
|
-
args={k: v for k, v in args.items() if k != "content"},
|
|
1017
|
-
)
|
|
1018
|
-
ctx.trace.tool("execute", name=name, outcome="blocked_approval", risk=risk)
|
|
1019
|
-
self._emit_step(ctx, "execute", "blocked", action=name, reason="approval")
|
|
1020
|
-
ctx.transcript.append({
|
|
1021
|
-
"state": AgentState.EXECUTING.value, "action": name,
|
|
1022
|
-
"thoughts": thoughts, "args": args, "risk": risk,
|
|
1023
|
-
"governance": dict(policy),
|
|
1024
|
-
"permission_mode": mode.value,
|
|
1025
|
-
"error": reason,
|
|
1026
|
-
})
|
|
1027
|
-
return True
|
|
1028
|
-
|
|
1029
|
-
def _dispatch_step(
|
|
1030
|
-
self, ctx: AgentRunContext, name: str, thoughts: str, args: dict,
|
|
1031
|
-
policy: Mapping[str, Any], risk: str, current_user: str,
|
|
1032
|
-
) -> None:
|
|
1033
|
-
"""Role check + shared tool lifecycle, recorded on the transcript either way."""
|
|
1034
|
-
d = self.deps
|
|
1035
|
-
sanitize_meta: Optional[Dict[str, Any]] = None
|
|
1036
|
-
if name == "write_file" and isinstance(args.get("content"), str):
|
|
1037
|
-
# ArtifactWritePipeline: the executor's args.content is untrusted
|
|
1038
|
-
# model output. The same extract→validate→repair guarantee as the
|
|
1039
|
-
# direct chat path applies here, so a weak model driving the JSON
|
|
1040
|
-
# loop can never persist fenced/chatty/truncated payloads.
|
|
1041
|
-
cleaned, meta = sanitize_write_content(
|
|
1042
|
-
str(args.get("path") or ""), args["content"],
|
|
1043
|
-
user_request=str(ctx.plan.get("goal") or thoughts or name),
|
|
1044
|
-
)
|
|
1045
|
-
if meta.get("sanitized"):
|
|
1046
|
-
args = dict(args)
|
|
1047
|
-
args["content"] = cleaned
|
|
1048
|
-
sanitize_meta = meta
|
|
1049
|
-
ctx.trace.repair(
|
|
1050
|
-
"execute",
|
|
1051
|
-
repairs=[
|
|
1052
|
-
"artifact_repair" if meta.get("repaired") else "artifact_sanitize"
|
|
1053
|
-
],
|
|
1054
|
-
)
|
|
1055
|
-
step_index = 1 + sum(
|
|
1056
|
-
1 for s in ctx.transcript
|
|
1057
|
-
if s.get("state") == AgentState.EXECUTING.value
|
|
1058
|
-
and s.get("action") not in (None, "final", "parse_error")
|
|
1059
|
-
)
|
|
1060
|
-
if (
|
|
1061
|
-
name in d.file_create_actions
|
|
1062
|
-
and d.snapshot_file is not None
|
|
1063
|
-
and args.get("path")
|
|
1064
|
-
):
|
|
1065
|
-
# Pre-write snapshot (review L7): the first capture per path is
|
|
1066
|
-
# the true pre-run state — later writes to the same path must
|
|
1067
|
-
# not overwrite it. Best-effort: a snapshot failure never
|
|
1068
|
-
# blocks the write, it only narrows rollback options.
|
|
1069
|
-
path_str = str(args["path"])
|
|
1070
|
-
if not any(entry.get("path") == path_str for entry in ctx.rollback_log):
|
|
1071
|
-
try:
|
|
1072
|
-
pre = d.snapshot_file(path_str)
|
|
1073
|
-
ctx.rollback_log.append({"path": path_str, **(pre or {})})
|
|
1074
|
-
except Exception as exc: # noqa: BLE001
|
|
1075
|
-
logging.warning("pre-write snapshot failed for %s: %s", path_str, exc)
|
|
1076
|
-
try:
|
|
1077
|
-
d.check_role(name, current_user)
|
|
1078
|
-
# Shared tool lifecycle: pre_tool (may block) → execute → post_tool.
|
|
1079
|
-
result = dispatch_tool(
|
|
1080
|
-
d.hooks, name, args,
|
|
1081
|
-
lambda: d.execute_tool(name, args),
|
|
1082
|
-
user_email=current_user, source="agent",
|
|
1083
|
-
)
|
|
1084
|
-
ctx.trace.tool("execute", name=name, outcome="ok", risk=risk)
|
|
1085
|
-
ctx.transcript.append({
|
|
1086
|
-
"state": AgentState.EXECUTING.value, "action": name,
|
|
1087
|
-
"thoughts": thoughts, "args": args,
|
|
1088
|
-
"risk": risk, "governance": dict(policy), "result": result,
|
|
1089
|
-
**({"content_sanitize": sanitize_meta} if sanitize_meta else {}),
|
|
1090
|
-
})
|
|
1091
|
-
self._emit_step(
|
|
1092
|
-
ctx, "execute", "tool", action=name, ok=True, step=step_index,
|
|
1093
|
-
path=str(args.get("path")) if args.get("path") else None,
|
|
1094
|
-
)
|
|
1095
|
-
except (ToolError, KeyError, TypeError, PermissionError) as exc:
|
|
1096
|
-
ctx.trace.tool("execute", name=name, outcome="error", risk=risk)
|
|
1097
|
-
ctx.transcript.append({
|
|
1098
|
-
"state": AgentState.EXECUTING.value, "action": name,
|
|
1099
|
-
"thoughts": thoughts, "args": args,
|
|
1100
|
-
"risk": risk, "governance": dict(policy), "error": str(exc),
|
|
1101
|
-
})
|
|
1102
|
-
self._emit_step(
|
|
1103
|
-
ctx, "execute", "tool", action=name, ok=False, step=step_index,
|
|
1104
|
-
path=str(args.get("path")) if args.get("path") else None,
|
|
1105
|
-
)
|
|
1106
|
-
|
|
1107
|
-
# ── VERIFY ───────────────────────────────────────────────────────
|
|
1108
|
-
def _has_execution_evidence(self, ctx: AgentRunContext) -> bool:
|
|
1109
|
-
"""Deterministic evidence check: at least one executing step actually
|
|
1110
|
-
produced a result (tool ran, or a governed change was staged as a
|
|
1111
|
-
proposal). ``final``/parse-error/blocked steps carry no result and do
|
|
1112
|
-
not count — a critic PASS over an evidence-free transcript must not
|
|
1113
|
-
become DONE."""
|
|
1114
|
-
for step in ctx.transcript:
|
|
1115
|
-
if step.get("state") != AgentState.EXECUTING.value:
|
|
1116
|
-
continue
|
|
1117
|
-
if step.get("action") in (None, "final", "parse_error"):
|
|
1118
|
-
continue
|
|
1119
|
-
if isinstance(step.get("result"), dict):
|
|
1120
|
-
return True
|
|
1121
|
-
return False
|
|
1122
|
-
|
|
1123
|
-
async def verify(
|
|
1124
|
-
self, ctx: AgentRunContext, req: Any, lang_hint: str, current_user: str,
|
|
1125
|
-
max_retry: int = 3, model_id: Optional[str] = None,
|
|
1126
|
-
) -> None:
|
|
1127
|
-
"""VERIFYING: Critic role evaluates transcript → DONE / EXECUTING (retry) / ROLLBACK / NEEDS_REVIEW / FAILED.
|
|
1128
|
-
|
|
1129
|
-
Fail-closed: a critic whose output cannot be parsed (after one strict
|
|
1130
|
-
repair retry) never fabricates a PASS — the run terminates as
|
|
1131
|
-
NEEDS_REVIEW so the user is told to check the result themselves.
|
|
1132
|
-
"""
|
|
1133
|
-
d = self.deps
|
|
1134
|
-
# The critic must see every step (evidence completeness), but not
|
|
1135
|
-
# every byte of tool output — long bodies are capped per string so
|
|
1136
|
-
# verification stays affordable on long runs (review Wave 0.3).
|
|
1137
|
-
verify_transcript = _truncate_strings(
|
|
1138
|
-
ctx.transcript, self.transcript_budget.verify_chars
|
|
1139
|
-
)
|
|
1140
|
-
# Deterministic artifact facts (review L4): the critic sees the
|
|
1141
|
-
# sanitize/repair honesty flags per written file, not just prose.
|
|
1142
|
-
checklist = artifact_checklist(ctx.transcript, d.file_create_actions)
|
|
1143
|
-
checklist_hint = (
|
|
1144
|
-
f"\n\n{format_artifact_checklist(checklist)}" if checklist else ""
|
|
1145
|
-
)
|
|
1146
|
-
# Requirement coverage (review 루프 §2): the critic previously judged
|
|
1147
|
-
# "did this fulfill the request?" from prose alone. It now also sees
|
|
1148
|
-
# which requested files actually exist and which requirements the user
|
|
1149
|
-
# spelled out.
|
|
1150
|
-
coverage = requirement_coverage(
|
|
1151
|
-
req.message, ctx.transcript, d.file_create_actions
|
|
1152
|
-
)
|
|
1153
|
-
context = (
|
|
1154
|
-
f"{d.critic_prompt}\n\n"
|
|
1155
|
-
f"[LANGUAGE HINT: {lang_hint}]\n\n"
|
|
1156
|
-
f"Original request: {req.message}\n"
|
|
1157
|
-
f"Plan goal: {ctx.plan.get('goal', req.message)}{checklist_hint}"
|
|
1158
|
-
f"{format_requirement_coverage(coverage)}\n\n"
|
|
1159
|
-
f"Full transcript:\n{json.dumps(verify_transcript, ensure_ascii=False, indent=2)}"
|
|
1160
|
-
)
|
|
1161
|
-
raw = await d.generate_as(
|
|
1162
|
-
model_id,
|
|
1163
|
-
message="Review the execution transcript and return your verdict JSON.",
|
|
1164
|
-
context=context, max_tokens=self.phase_budgets.verify_tokens, temperature=0.1,
|
|
1165
|
-
)
|
|
1166
|
-
ctx.trace.llm_call("verify", model=model_id)
|
|
1167
|
-
verdict: Optional[Dict[str, Any]] = None
|
|
1168
|
-
try:
|
|
1169
|
-
verdict, verdict_repairs = extract_action_details(str(raw))
|
|
1170
|
-
ctx.trace.repair("verify", repairs=verdict_repairs)
|
|
1171
|
-
except ValueError as exc:
|
|
1172
|
-
# One strict repair retry — re-ask the critic for the exact wire
|
|
1173
|
-
# format instead of fabricating a verdict.
|
|
1174
|
-
ctx.trace.parse_error("verify", error=str(exc), recovered=True)
|
|
1175
|
-
strict_context = (
|
|
1176
|
-
f"{context}\n\n"
|
|
1177
|
-
"Your previous verdict was not parseable JSON. Reply with EXACTLY one "
|
|
1178
|
-
'JSON object like {"action": "verdict", "verdict": "PASS", '
|
|
1179
|
-
'"next_state": "DONE", "reason": "...", "corrections": []} '
|
|
1180
|
-
"and nothing else. verdict must be PASS or FAIL; next_state must be "
|
|
1181
|
-
"one of DONE, EXECUTING, ROLLBACK, FAILED."
|
|
1182
|
-
)
|
|
1183
|
-
raw = await d.generate_as(
|
|
1184
|
-
model_id,
|
|
1185
|
-
message="Return your verdict as one strict JSON object.",
|
|
1186
|
-
context=strict_context, max_tokens=self.phase_budgets.verify_tokens,
|
|
1187
|
-
temperature=0.0,
|
|
1188
|
-
)
|
|
1189
|
-
ctx.trace.llm_call("verify", model=model_id)
|
|
1190
|
-
try:
|
|
1191
|
-
verdict, verdict_repairs = extract_action_details(str(raw))
|
|
1192
|
-
ctx.trace.repair("verify", repairs=verdict_repairs)
|
|
1193
|
-
except ValueError as retry_exc:
|
|
1194
|
-
ctx.trace.parse_error("verify", error=str(retry_exc), recovered=False)
|
|
1195
|
-
verdict = None
|
|
1196
|
-
|
|
1197
|
-
has_evidence = self._has_execution_evidence(ctx)
|
|
1198
|
-
|
|
1199
|
-
if verdict is None:
|
|
1200
|
-
# Verifier unavailable — fail closed, never DONE.
|
|
1201
|
-
ctx.transcript.append({
|
|
1202
|
-
"state": AgentState.VERIFYING.value,
|
|
1203
|
-
"verdict": "UNAVAILABLE",
|
|
1204
|
-
"reason": "critic output unparseable after strict retry",
|
|
1205
|
-
"verifier_available": False,
|
|
1206
|
-
"verdict_valid": False,
|
|
1207
|
-
"evidence": has_evidence,
|
|
1208
|
-
})
|
|
1209
|
-
ctx.trace.decision(
|
|
1210
|
-
"verify", decision="verification_unavailable",
|
|
1211
|
-
verifier_available=False, verdict_valid=False, evidence=has_evidence,
|
|
1212
|
-
)
|
|
1213
|
-
self._emit_step(ctx, "verify", "verdict", verdict="UNAVAILABLE")
|
|
1214
|
-
ctx.final_message = (
|
|
1215
|
-
"검증을 완료하지 못했습니다 — 검증 모델의 응답을 해석할 수 없었습니다. "
|
|
1216
|
-
"실행 결과를 직접 확인해 주시고, 필요하면 다시 시도해 주세요."
|
|
1217
|
-
)
|
|
1218
|
-
ctx.state = AgentState.NEEDS_REVIEW
|
|
1219
|
-
return
|
|
1220
|
-
|
|
1221
|
-
ctx.corrections = verdict.get("corrections", [])
|
|
1222
|
-
# Normalize legacy verdict next_state strings to current AgentState names
|
|
1223
|
-
raw_next = verdict.get("next_state", "")
|
|
1224
|
-
next_s = {"COMPLETE": "DONE", "RETRY": "EXECUTING"}.get(raw_next, raw_next)
|
|
1225
|
-
|
|
1226
|
-
ctx.transcript.append({
|
|
1227
|
-
"state": AgentState.VERIFYING.value,
|
|
1228
|
-
"verdict": verdict.get("verdict", ""),
|
|
1229
|
-
"reason": verdict.get("reason", ""),
|
|
1230
|
-
"corrections": ctx.corrections,
|
|
1231
|
-
"confidence": verdict.get("confidence", 0.9),
|
|
1232
|
-
"next_state": next_s,
|
|
1233
|
-
"verifier_available": True,
|
|
1234
|
-
"verdict_valid": True,
|
|
1235
|
-
"evidence": has_evidence,
|
|
1236
|
-
})
|
|
1237
|
-
|
|
1238
|
-
ctx.trace.decision(
|
|
1239
|
-
"verify", decision=str(verdict.get("verdict", "")), next_state=next_s,
|
|
1240
|
-
verifier_available=True, verdict_valid=True, evidence=has_evidence,
|
|
1241
|
-
)
|
|
1242
|
-
self._emit_step(
|
|
1243
|
-
ctx, "verify", "verdict",
|
|
1244
|
-
verdict=str(verdict.get("verdict", "")), next_state=next_s,
|
|
1245
|
-
)
|
|
1246
|
-
if verdict.get("verdict") == "PASS":
|
|
1247
|
-
# DONE requires both: a validly parsed PASS verdict AND
|
|
1248
|
-
# deterministic execution evidence in the transcript. A PASS over
|
|
1249
|
-
# an evidence-free run is not a completion.
|
|
1250
|
-
if not has_evidence:
|
|
1251
|
-
ctx.trace.decision("verify", decision="needs_review_no_evidence")
|
|
1252
|
-
ctx.final_message = (
|
|
1253
|
-
"검증자는 통과를 보고했지만 실제 실행 근거(도구 실행 기록)가 없어 "
|
|
1254
|
-
"완료로 처리하지 않았습니다. 결과를 직접 확인해 주세요."
|
|
1255
|
-
)
|
|
1256
|
-
ctx.state = AgentState.NEEDS_REVIEW
|
|
1257
|
-
return
|
|
1258
|
-
if not coverage["complete"]:
|
|
1259
|
-
# A PASS that leaves a *requested file* unwritten is not a
|
|
1260
|
-
# completion — this is a fact, not a judgement, so it is
|
|
1261
|
-
# enforced rather than merely reported to the critic.
|
|
1262
|
-
missing = ", ".join(coverage["missing_files"])
|
|
1263
|
-
ctx.trace.decision(
|
|
1264
|
-
"verify", decision="needs_review_missing_files",
|
|
1265
|
-
missing=len(coverage["missing_files"]),
|
|
1266
|
-
)
|
|
1267
|
-
ctx.transcript.append({
|
|
1268
|
-
"state": AgentState.VERIFYING.value,
|
|
1269
|
-
"requirement_coverage": coverage,
|
|
1270
|
-
})
|
|
1271
|
-
ctx.final_message = (
|
|
1272
|
-
f"요청한 파일 중 일부가 만들어지지 않아 완료로 처리하지 않았습니다: {missing}"
|
|
1273
|
-
)
|
|
1274
|
-
ctx.state = AgentState.NEEDS_REVIEW
|
|
1275
|
-
return
|
|
1276
|
-
if not ctx.final_message:
|
|
1277
|
-
ctx.final_message = verdict.get("reason", "작업이 완료되었습니다.")
|
|
1278
|
-
ctx.state = AgentState.DONE
|
|
1279
|
-
elif next_s == "ROLLBACK":
|
|
1280
|
-
ctx.state = AgentState.ROLLBACK
|
|
1281
|
-
elif next_s == "EXECUTING":
|
|
1282
|
-
if ctx.retry_count >= max_retry:
|
|
1283
|
-
ctx.final_message = "처리 중 문제가 발생했습니다. 다시 시도해 주세요."
|
|
1284
|
-
ctx.state = AgentState.FAILED
|
|
1285
|
-
else:
|
|
1286
|
-
ctx.retry_count += 1
|
|
1287
|
-
ctx.trace.retry("verify", attempt=ctx.retry_count)
|
|
1288
|
-
ctx.transcript.append({
|
|
1289
|
-
"state": AgentState.EXECUTING.value,
|
|
1290
|
-
"retry_attempt": ctx.retry_count,
|
|
1291
|
-
"corrections": ctx.corrections,
|
|
1292
|
-
})
|
|
1293
|
-
ctx.state = AgentState.EXECUTING
|
|
1294
|
-
elif next_s == "DONE":
|
|
1295
|
-
# Contradictory verdict: the critic asked for DONE without a PASS.
|
|
1296
|
-
# The loose "or next_state == DONE" success path is gone — this is
|
|
1297
|
-
# a non-success that the user must review.
|
|
1298
|
-
ctx.trace.decision("verify", decision="needs_review_inconsistent_verdict")
|
|
1299
|
-
ctx.final_message = (
|
|
1300
|
-
"검증 결과가 일관되지 않아 완료로 처리하지 않았습니다. "
|
|
1301
|
-
"실행 결과를 직접 확인해 주세요."
|
|
1302
|
-
)
|
|
1303
|
-
ctx.state = AgentState.NEEDS_REVIEW
|
|
1304
|
-
else:
|
|
1305
|
-
ctx.final_message = verdict.get("reason", "검증자가 인식되지 않은 다음 상태를 반환했습니다.")
|
|
1306
|
-
ctx.state = AgentState.FAILED
|
|
1307
|
-
|
|
1308
|
-
# ── ROLLBACK ─────────────────────────────────────────────────────
|
|
1309
|
-
def _snapshot_for(self, ctx: AgentRunContext, path: str) -> Optional[Dict[str, Any]]:
|
|
1310
|
-
for entry in ctx.rollback_log:
|
|
1311
|
-
if entry.get("path") == path:
|
|
1312
|
-
return entry
|
|
1313
|
-
return None
|
|
1314
|
-
|
|
1315
|
-
def _rollback_one(self, ctx: AgentRunContext, path: str, gov: Dict[str, Any]) -> Dict[str, Any]:
|
|
1316
|
-
"""Recover one path: git when governed and available, else the
|
|
1317
|
-
pre-write snapshot, else an honest ``mode="none"`` (review L7)."""
|
|
1318
|
-
d = self.deps
|
|
1319
|
-
if gov.get("rollback") == "git" and d.rollback_file is not None:
|
|
1320
|
-
try:
|
|
1321
|
-
result = dict(d.rollback_file(str(path)))
|
|
1322
|
-
except Exception as exc: # noqa: BLE001
|
|
1323
|
-
result = {"path": path, "ok": False, "error": str(exc)}
|
|
1324
|
-
if result.get("ok"):
|
|
1325
|
-
result["mode"] = "git"
|
|
1326
|
-
return result
|
|
1327
|
-
snapshot = self._snapshot_for(ctx, str(path))
|
|
1328
|
-
if snapshot is not None and d.restore_snapshot is not None and not snapshot.get("too_large"):
|
|
1329
|
-
content = snapshot.get("content") if snapshot.get("existed") else None
|
|
1330
|
-
try:
|
|
1331
|
-
restored = dict(d.restore_snapshot(str(path), content))
|
|
1332
|
-
except Exception as exc: # noqa: BLE001
|
|
1333
|
-
restored = {"path": path, "ok": False, "error": str(exc)}
|
|
1334
|
-
restored.setdefault("path", path)
|
|
1335
|
-
restored["mode"] = "snapshot"
|
|
1336
|
-
return restored
|
|
1337
|
-
return {
|
|
1338
|
-
"path": path, "ok": False, "mode": "none",
|
|
1339
|
-
"error": "no rollback available (git not applicable, no usable snapshot)",
|
|
1340
|
-
}
|
|
1341
|
-
|
|
1342
|
-
def rollback(self, ctx: AgentRunContext, current_user: str) -> None:
|
|
1343
|
-
"""ROLLBACK: recover written files (git → snapshot → none), then FAILED."""
|
|
1344
|
-
d = self.deps
|
|
1345
|
-
rolled: List[dict] = []
|
|
1346
|
-
seen_paths: set = set()
|
|
1347
|
-
for step in ctx.transcript:
|
|
1348
|
-
if step.get("state") != AgentState.EXECUTING.value:
|
|
1349
|
-
continue
|
|
1350
|
-
if not isinstance(step.get("result"), dict):
|
|
1351
|
-
continue
|
|
1352
|
-
gov = step.get("governance", {}) or {}
|
|
1353
|
-
path = step["result"].get("path") or (step.get("args") or {}).get("path", "")
|
|
1354
|
-
if not path or str(path) in seen_paths:
|
|
1355
|
-
continue
|
|
1356
|
-
if gov.get("rollback") != "git" and step.get("action") not in d.file_create_actions:
|
|
1357
|
-
continue
|
|
1358
|
-
seen_paths.add(str(path))
|
|
1359
|
-
rolled.append(self._rollback_one(ctx, str(path), gov))
|
|
1360
|
-
|
|
1361
|
-
ctx.transcript.append({"state": AgentState.ROLLBACK.value, "rolled_back": rolled})
|
|
1362
|
-
ctx.trace.decision(
|
|
1363
|
-
"rollback", decision="rolled_back",
|
|
1364
|
-
attempted=len(rolled), recovered=sum(1 for r in rolled if r.get("ok")),
|
|
1365
|
-
)
|
|
1366
|
-
recovered = [f"{r['path']} ({r.get('mode')})" for r in rolled if r.get("ok")]
|
|
1367
|
-
ctx.final_message = (
|
|
1368
|
-
f"실행 실패로 롤백했습니다. 복구 파일: {recovered}"
|
|
1369
|
-
if recovered
|
|
1370
|
-
else "롤백을 시도했으나 복구할 파일이 없거나 git/스냅샷 복구 수단이 없습니다."
|
|
1371
|
-
)
|
|
1372
|
-
d.audit("agent_rollback", user_email=current_user, rolled_back=rolled)
|
|
1373
|
-
self._emit_step(ctx, "rollback", "rolled_back", recovered=len(recovered))
|
|
1374
|
-
# Rollback is a recovery from a failed verification — terminal state is FAILED
|
|
1375
|
-
ctx.state = AgentState.FAILED
|
|
1376
|
-
|
|
1377
|
-
# ── MEMORY ───────────────────────────────────────────────────────
|
|
1378
|
-
async def memory_update(self, ctx: AgentRunContext, req: Any, current_user: str) -> None:
|
|
1379
|
-
"""Background: Memory Updater role extracts learnings from a terminal run.
|
|
1380
|
-
|
|
1381
|
-
Terminal-state learning policy (review §4.2 L6): DONE runs record what
|
|
1382
|
-
worked; FAILED / NEEDS_REVIEW runs record what went wrong — failure is
|
|
1383
|
-
exactly the experience worth remembering. The run status stored with
|
|
1384
|
-
the experience is the *actual* terminal state, never a blanket "ok".
|
|
1385
|
-
"""
|
|
1386
|
-
d = self.deps
|
|
1387
|
-
terminal = ctx.state.value if ctx.state in AGENT_TERMINAL_STATES else "UNKNOWN"
|
|
1388
|
-
outcome_hint = (
|
|
1389
|
-
"The task completed successfully."
|
|
1390
|
-
if ctx.state == AgentState.DONE
|
|
1391
|
-
else (
|
|
1392
|
-
f"The task ended as {terminal} — extract what went wrong and "
|
|
1393
|
-
"what to do differently next time, not a success story."
|
|
1394
|
-
)
|
|
1395
|
-
)
|
|
1396
|
-
context = (
|
|
1397
|
-
f"{d.memory_updater_prompt}\n\n"
|
|
1398
|
-
f"Task: {req.message}\n"
|
|
1399
|
-
f"Terminal status: {terminal}. {outcome_hint}\n\n"
|
|
1400
|
-
f"Last 5 transcript steps:\n{json.dumps(ctx.transcript[-5:], ensure_ascii=False)}"
|
|
1401
|
-
)
|
|
1402
|
-
try:
|
|
1403
|
-
raw = await d.generate(
|
|
1404
|
-
message="Extract learnings from this completed task.",
|
|
1405
|
-
context=context, max_tokens=self.phase_budgets.memory_tokens, temperature=0.1,
|
|
1406
|
-
)
|
|
1407
|
-
mem = extract_action(str(raw))
|
|
1408
|
-
kept_learnings = filter_learnings(mem.get("learnings") or [])
|
|
1409
|
-
if mem.get("save_to_knowledge") and kept_learnings:
|
|
1410
|
-
learnings = "\n".join(kept_learnings)
|
|
1411
|
-
status_label = {
|
|
1412
|
-
AgentState.DONE: "ok",
|
|
1413
|
-
AgentState.NEEDS_REVIEW: "needs_review",
|
|
1414
|
-
AgentState.FAILED: "failed",
|
|
1415
|
-
}.get(ctx.state, "unknown")
|
|
1416
|
-
if d.brain_memory is not None:
|
|
1417
|
-
# This runtime is LLM-driven — its learnings are real
|
|
1418
|
-
# experiences and enter the brain with provenance.
|
|
1419
|
-
d.brain_memory.record_experience(
|
|
1420
|
-
f"Agent: {req.message[:60]}",
|
|
1421
|
-
learnings,
|
|
1422
|
-
run={
|
|
1423
|
-
"mode": "llm",
|
|
1424
|
-
"status": status_label,
|
|
1425
|
-
"agent_id": "agent:executor",
|
|
1426
|
-
"steps": len(ctx.transcript),
|
|
1427
|
-
},
|
|
1428
|
-
user_email=current_user or None,
|
|
1429
|
-
)
|
|
1430
|
-
else:
|
|
1431
|
-
d.knowledge_save(
|
|
1432
|
-
learnings,
|
|
1433
|
-
folder="30_Projects",
|
|
1434
|
-
title=f"Agent: {req.message[:60]}",
|
|
1435
|
-
)
|
|
1436
|
-
except Exception as exc:
|
|
1437
|
-
# Never crash a completed run, but never swallow silently either.
|
|
1438
|
-
logging.warning("agent memory update failed: %s", exc)
|
|
1439
|
-
|
|
1440
|
-
# ── DRIVE LOOP ───────────────────────────────────────────────────
|
|
1441
|
-
async def run_to_completion(
|
|
1442
|
-
self, ctx: AgentRunContext, req: Any, lang_hint: str,
|
|
1443
|
-
current_user: str, max_steps: int, max_retry: int,
|
|
1444
|
-
) -> None:
|
|
1445
|
-
"""Run EXECUTING → VERIFYING → ROLLBACK loop until a terminal state."""
|
|
1446
|
-
while ctx.state not in AGENT_TERMINAL_STATES:
|
|
1447
|
-
ctx.state_history.append(ctx.state.value)
|
|
1448
|
-
if len(ctx.state_history) > 200:
|
|
1449
|
-
ctx.final_message = "에이전트 상태 머신이 최대 반복(200)에 도달해 중단했습니다."
|
|
1450
|
-
ctx.state = AgentState.FAILED
|
|
1451
|
-
break
|
|
1452
|
-
|
|
1453
|
-
if ctx.state == AgentState.EXECUTING:
|
|
1454
|
-
await self.execute(ctx, req, lang_hint, current_user, max_steps,
|
|
1455
|
-
model_id=ctx.executing_model)
|
|
1456
|
-
elif ctx.state == AgentState.VERIFYING:
|
|
1457
|
-
await self.verify(ctx, req, lang_hint, current_user, max_retry,
|
|
1458
|
-
model_id=ctx.reviewing_model)
|
|
1459
|
-
elif ctx.state == AgentState.ROLLBACK:
|
|
1460
|
-
self.rollback(ctx, current_user)
|
|
1461
|
-
else:
|
|
1462
|
-
ctx.state = AgentState.FAILED
|
|
1463
|
-
|
|
1464
|
-
ctx.state_history.append(ctx.state.value)
|
|
1465
|
-
self._emit_step(ctx, "terminal", "state", state=ctx.state.value)
|