deer-agent-framework 0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- deer/__init__.py +36 -0
- deer/builtins/__init__.py +7 -0
- deer/builtins/python_manager/agent.py +29 -0
- deer/builtins/python_manager/tools.py +54 -0
- deer/core/__init__.py +1 -0
- deer/core/agent.py +463 -0
- deer/core/ui.py +31 -0
- deer/drivers/__init__.py +66 -0
- deer/drivers/base_driver.py +56 -0
- deer/drivers/gemini_driver.py +62 -0
- deer/drivers/ollama_driver.py +69 -0
- deer/executor/__init__.py +1 -0
- deer/executor/executor.py +168 -0
- deer/executor/logic.py +75 -0
- deer/executor/logic_secure.py +102 -0
- deer/main.py +71 -0
- deer/planner/__init__.py +1 -0
- deer/planner/planner.py +71 -0
- deer/prompts/__init__.py +6 -0
- deer/prompts/error_explain.py +30 -0
- deer/prompts/goal_improvement.py +65 -0
- deer/prompts/goal_validation.py +31 -0
- deer/prompts/humanizer.py +20 -0
- deer/prompts/planner.py +92 -0
- deer/prompts/response_improvement.py +23 -0
- deer/schema/__init__.py +2 -0
- deer/schema/io.py +40 -0
- deer/schema/plan.py +75 -0
- deer/tools/__init__.py +4 -0
- deer/tools/base.py +141 -0
- deer/tools/builtin/__init__.py +3 -0
- deer/tools/builtin/file_manager.py +136 -0
- deer/tools/builtin/git_manager.py +67 -0
- deer/tools/builtin/search_manager.py +123 -0
- deer/tools/decorators.py +127 -0
- deer/tools/registry.py +114 -0
- deer/tracing/__init__.py +2 -0
- deer/tracing/logging_config.py +20 -0
- deer/tracing/store.py +21 -0
- deer/utils/__init__.py +0 -0
- deer/utils/console.py +11 -0
- deer/utils/plots/__init__.py +7 -0
- deer/utils/plots/plot_traces.py +1034 -0
- deer/validator/__init__.py +1 -0
- deer/validator/plan_validator.py +23 -0
- deer/validator/rules.py +98 -0
- deer_agent_framework-0.0.dist-info/METADATA +163 -0
- deer_agent_framework-0.0.dist-info/RECORD +52 -0
- deer_agent_framework-0.0.dist-info/WHEEL +5 -0
- deer_agent_framework-0.0.dist-info/entry_points.txt +2 -0
- deer_agent_framework-0.0.dist-info/licenses/LICENSE +24 -0
- deer_agent_framework-0.0.dist-info/top_level.txt +1 -0
deer/__init__.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import os
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def load_env_file(env_path: Path) -> None:
|
|
6
|
+
"""Reads a .env file and loads its variables into os.environ natively."""
|
|
7
|
+
try:
|
|
8
|
+
with open(env_path, "r", encoding="utf-8") as f:
|
|
9
|
+
for line in f:
|
|
10
|
+
# Strip whitespace and ignore empty lines or comments
|
|
11
|
+
line = line.strip()
|
|
12
|
+
if not line or line.startswith("#"):
|
|
13
|
+
continue
|
|
14
|
+
|
|
15
|
+
# Split by the first '=' found
|
|
16
|
+
if "=" in line:
|
|
17
|
+
key, value = line.split("=", 1)
|
|
18
|
+
# Strip quotes if the value is wrapped in them
|
|
19
|
+
key = key.strip()
|
|
20
|
+
value = value.strip().strip("'\"")
|
|
21
|
+
|
|
22
|
+
# Set the environment variable
|
|
23
|
+
os.environ[key] = value
|
|
24
|
+
except IOError as e:
|
|
25
|
+
# Log or handle the error if the file exists but can't be read
|
|
26
|
+
pass
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
# Your configuration logic remains clean and dependency-free
|
|
30
|
+
local_env = Path(".env")
|
|
31
|
+
user_env = Path.home() / "deer.env"
|
|
32
|
+
|
|
33
|
+
if local_env.exists():
|
|
34
|
+
load_env_file(local_env)
|
|
35
|
+
elif user_env.exists():
|
|
36
|
+
load_env_file(user_env)
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
|
|
3
|
+
from deer.core.agent import DeterministicAgent
|
|
4
|
+
from deer.drivers import get_driver_from_parser
|
|
5
|
+
|
|
6
|
+
if __package__:
|
|
7
|
+
from .tools import registry
|
|
8
|
+
else:
|
|
9
|
+
from tools import registry
|
|
10
|
+
|
|
11
|
+
agent = DeterministicAgent(
|
|
12
|
+
description="Python Architecture Specialist expert in module lifecycle, dependency management, and package distribution.",
|
|
13
|
+
identity=(
|
|
14
|
+
"You are a Principal Python Architect and Core Ecosystem Specialist. You possess authoritative expertise "
|
|
15
|
+
"in advanced module resolution, dependency management architectures, package distribution, and internal "
|
|
16
|
+
"or public repository management. Your knowledge spans the entire lifecycle of Python code: from how "
|
|
17
|
+
"modules are imported and structured at runtime, to how dependencies are resolved, isolated, packaged, "
|
|
18
|
+
"and deployed across diverse environments."
|
|
19
|
+
),
|
|
20
|
+
driver=get_driver_from_parser(),
|
|
21
|
+
registry=registry,
|
|
22
|
+
jail_path=Path.cwd(),
|
|
23
|
+
format_response="markdown",
|
|
24
|
+
max_tries_for_plan=5,
|
|
25
|
+
rollback=None,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
if __name__ == "__main__":
|
|
29
|
+
agent.repl()
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
from deer.tools.registry import ToolRegistry
|
|
2
|
+
from deer.tools.builtin import FileManager, GitManager, SearchManager
|
|
3
|
+
|
|
4
|
+
jail_path = "/Users/yeison/Development/deer-agent-framework/sandbox/root"
|
|
5
|
+
|
|
6
|
+
registry = ToolRegistry()
|
|
7
|
+
|
|
8
|
+
fileManager = FileManager(
|
|
9
|
+
# tools=[
|
|
10
|
+
# "new_file",
|
|
11
|
+
# "read_file",
|
|
12
|
+
# "delete_file",
|
|
13
|
+
# "create_directory",
|
|
14
|
+
# "get_file_info",
|
|
15
|
+
# "directory_tree",
|
|
16
|
+
# ],
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
gitManager = GitManager(
|
|
20
|
+
# tools=[
|
|
21
|
+
# "git_status",
|
|
22
|
+
# "git_current_branch",
|
|
23
|
+
# "git_log",
|
|
24
|
+
# "git_diff",
|
|
25
|
+
# "git_staged_diff",
|
|
26
|
+
# "git_show",
|
|
27
|
+
# "git_add",
|
|
28
|
+
# "git_commit",
|
|
29
|
+
# "git_restore",
|
|
30
|
+
# ],
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
searchManager = SearchManager(
|
|
34
|
+
# tools=[
|
|
35
|
+
# "search_text",
|
|
36
|
+
# "search_regex",
|
|
37
|
+
# "search_text_ignore_case",
|
|
38
|
+
# "search_regex_ignore_case",
|
|
39
|
+
# "find_files",
|
|
40
|
+
# "search_file_names",
|
|
41
|
+
# "list_files",
|
|
42
|
+
# "search_by_extension",
|
|
43
|
+
# "search_text_in_files",
|
|
44
|
+
# "files_with_matches",
|
|
45
|
+
# "files_without_matches",
|
|
46
|
+
# "count_matches",
|
|
47
|
+
# ],
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
registry.register(
|
|
51
|
+
fileManager,
|
|
52
|
+
gitManager,
|
|
53
|
+
searchManager,
|
|
54
|
+
)
|
deer/core/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
from .agent import DeterministicAgent
|
deer/core/agent.py
ADDED
|
@@ -0,0 +1,463 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import json
|
|
3
|
+
import re
|
|
4
|
+
import pickle
|
|
5
|
+
import sys
|
|
6
|
+
from typing import Callable
|
|
7
|
+
from datetime import datetime
|
|
8
|
+
|
|
9
|
+
from prompt_toolkit import prompt
|
|
10
|
+
from prompt_toolkit.formatted_text import HTML
|
|
11
|
+
from prompt_toolkit.history import InMemoryHistory
|
|
12
|
+
|
|
13
|
+
from deer.prompts import (
|
|
14
|
+
RESPONSE_IMPROVEMENT_PROMPT,
|
|
15
|
+
ERROR_EXPLAIN_PROMPT,
|
|
16
|
+
HUMANIZER_PROMPT,
|
|
17
|
+
GOAL_VERIFIER_PROMPT,
|
|
18
|
+
VERIFIER_JUDGE_PROMPT,
|
|
19
|
+
)
|
|
20
|
+
from deer.drivers.base_driver import LLMDriver
|
|
21
|
+
from deer.planner.planner import Planner
|
|
22
|
+
from deer.validator.plan_validator import PlanValidator
|
|
23
|
+
from deer.executor.executor import Executor
|
|
24
|
+
from deer.core.ui import WELCOME_MESSAGE
|
|
25
|
+
from deer.tools.registry import ToolRegistry, default_registry
|
|
26
|
+
from deer.schema.io import AgentInput, AgentOutput
|
|
27
|
+
|
|
28
|
+
from rich.console import Console
|
|
29
|
+
from rich.markdown import Markdown
|
|
30
|
+
|
|
31
|
+
logger = logging.getLogger("DEER")
|
|
32
|
+
prompt_history = InMemoryHistory()
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class DeterministicAgent:
|
|
36
|
+
def __init__(
|
|
37
|
+
self,
|
|
38
|
+
description: str = "",
|
|
39
|
+
identity: str = "DeterministicAgent",
|
|
40
|
+
max_tries_for_plan: int = 3,
|
|
41
|
+
driver: LLMDriver | None = None,
|
|
42
|
+
registry: ToolRegistry | None = None,
|
|
43
|
+
format_response: str = "plaintext",
|
|
44
|
+
rollback: Callable = None,
|
|
45
|
+
jail_path: str = None,
|
|
46
|
+
) -> None:
|
|
47
|
+
assert format_response in {"markdown", "plaintext"}
|
|
48
|
+
|
|
49
|
+
self.identity = identity
|
|
50
|
+
self.description = description
|
|
51
|
+
self.driver = driver
|
|
52
|
+
self.registry = registry or default_registry()
|
|
53
|
+
self.console = Console()
|
|
54
|
+
|
|
55
|
+
self.executor = Executor(
|
|
56
|
+
registry=self.registry,
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
self.validator = PlanValidator(
|
|
60
|
+
registry=self.registry,
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
self.planner = Planner(
|
|
64
|
+
identity=identity,
|
|
65
|
+
driver=driver,
|
|
66
|
+
registry=self.registry,
|
|
67
|
+
format_response=format_response,
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
assert (
|
|
71
|
+
max_tries_for_plan > 0
|
|
72
|
+
), f"max_tries_for_plan must be positive, got {max_tries_for_plan}"
|
|
73
|
+
self.max_tries_for_plan = max_tries_for_plan
|
|
74
|
+
|
|
75
|
+
self.history = []
|
|
76
|
+
self.trace = []
|
|
77
|
+
self.rollback_function = rollback
|
|
78
|
+
|
|
79
|
+
if jail_path:
|
|
80
|
+
self.set_jail(jail_path)
|
|
81
|
+
|
|
82
|
+
logger.info(f"DEER - Deterministic Executable Engine for Runtime-agents")
|
|
83
|
+
logger.info(
|
|
84
|
+
f"Initialized DeterministicAgent with '{self.driver.model_name}' model"
|
|
85
|
+
)
|
|
86
|
+
logger.info(f"Identity: {self.identity}")
|
|
87
|
+
logger.info(
|
|
88
|
+
f"Available tools:\n{self.registry.describe(include_state_modifying=True)}"
|
|
89
|
+
)
|
|
90
|
+
logger.info(f"Authorized filesystem scope is restricted to: {jail_path}")
|
|
91
|
+
|
|
92
|
+
def set_jail(self, jail_path):
|
|
93
|
+
self.registry.set_jail(jail_path)
|
|
94
|
+
|
|
95
|
+
def should_verify(self, plan) -> bool:
|
|
96
|
+
return any(step.tool for step in plan.steps)
|
|
97
|
+
|
|
98
|
+
def execute_plan(
|
|
99
|
+
self,
|
|
100
|
+
agent_input: AgentInput,
|
|
101
|
+
feedback: {},
|
|
102
|
+
include_state_modifying: bool = True,
|
|
103
|
+
):
|
|
104
|
+
last_error_message = ""
|
|
105
|
+
|
|
106
|
+
try:
|
|
107
|
+
plan = self.planner.plan(
|
|
108
|
+
agent_input,
|
|
109
|
+
feedback,
|
|
110
|
+
include_state_modifying=include_state_modifying,
|
|
111
|
+
)
|
|
112
|
+
except Exception as planning_error:
|
|
113
|
+
logger.warning(f"Planning error: {planning_error}")
|
|
114
|
+
feedback = {"Planning error": str(planning_error)}
|
|
115
|
+
last_error_message = str(planning_error)
|
|
116
|
+
return None, feedback, last_error_message, None
|
|
117
|
+
logger.info(f"Plan generated")
|
|
118
|
+
logger.debug(f"Plan steps:")
|
|
119
|
+
for step in plan.steps:
|
|
120
|
+
logger.debug(f"\tstep: {step.id}")
|
|
121
|
+
if step.tool:
|
|
122
|
+
logger.debug(f"\t\ttool: {step.tool}: {step.params}")
|
|
123
|
+
elif step.logic:
|
|
124
|
+
logger.debug(f"\t\tlogic: {step.logic}")
|
|
125
|
+
|
|
126
|
+
try:
|
|
127
|
+
self.validator.validate(plan)
|
|
128
|
+
except Exception as validation_error:
|
|
129
|
+
logger.warning(f"Verification error: {validation_error}")
|
|
130
|
+
feedback = {"verification error": str(validation_error)}
|
|
131
|
+
last_error_message = str(validation_error)
|
|
132
|
+
return None, feedback, last_error_message, plan
|
|
133
|
+
logger.info(f"Plan verified")
|
|
134
|
+
|
|
135
|
+
try:
|
|
136
|
+
response = self.executor.execute(
|
|
137
|
+
plan,
|
|
138
|
+
agent_input.payload,
|
|
139
|
+
)
|
|
140
|
+
except Exception as execution_error:
|
|
141
|
+
logger.warning(f"Execution error: {execution_error}")
|
|
142
|
+
feedback = {"execution error": str(execution_error)}
|
|
143
|
+
last_error_message = str(execution_error)
|
|
144
|
+
return None, feedback, last_error_message, plan
|
|
145
|
+
logger.info(f"Plan executed")
|
|
146
|
+
|
|
147
|
+
return response, feedback, last_error_message, plan
|
|
148
|
+
|
|
149
|
+
def run(self, agent_input: AgentInput) -> AgentOutput:
|
|
150
|
+
feedback = {}
|
|
151
|
+
run_trace = {}
|
|
152
|
+
|
|
153
|
+
response_validation = None
|
|
154
|
+
|
|
155
|
+
for attempt_idx in range(self.max_tries_for_plan):
|
|
156
|
+
logger.debug(f"Attempt {attempt_idx + 1} of {self.max_tries_for_plan}")
|
|
157
|
+
|
|
158
|
+
run_trace[f"Attempt-{attempt_idx + 1}"] = {}
|
|
159
|
+
|
|
160
|
+
for solution_idx in range(self.max_tries_for_plan):
|
|
161
|
+
logger.debug(
|
|
162
|
+
f"Solution planning {solution_idx + 1} of {self.max_tries_for_plan}"
|
|
163
|
+
)
|
|
164
|
+
run_trace[f"Attempt-{attempt_idx + 1}"][
|
|
165
|
+
f"Solution-{solution_idx + 1}"
|
|
166
|
+
] = datetime.now()
|
|
167
|
+
|
|
168
|
+
response, feedback, last_error_message, plan = self.execute_plan(
|
|
169
|
+
agent_input, feedback
|
|
170
|
+
)
|
|
171
|
+
if response is None:
|
|
172
|
+
continue
|
|
173
|
+
break
|
|
174
|
+
|
|
175
|
+
if response is None:
|
|
176
|
+
logger.warning(
|
|
177
|
+
f"Agent failed to produce a response after {self.max_tries_for_plan} attempts"
|
|
178
|
+
)
|
|
179
|
+
if last_error_message:
|
|
180
|
+
last_error_message_explained = self.explain_error(
|
|
181
|
+
last_error_message
|
|
182
|
+
)
|
|
183
|
+
return AgentOutput(
|
|
184
|
+
result=last_error_message_explained,
|
|
185
|
+
trace=self.executor.trace_store.get_trace(),
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
if not self.should_verify(plan):
|
|
189
|
+
break
|
|
190
|
+
|
|
191
|
+
feedback_ = {}
|
|
192
|
+
goal_verifier_prompt = GOAL_VERIFIER_PROMPT.format(goal=agent_input.goal)
|
|
193
|
+
|
|
194
|
+
payload_verifier = {
|
|
195
|
+
"original_goal": agent_input.goal,
|
|
196
|
+
"claimed_result": response.result,
|
|
197
|
+
"execution_trace": [
|
|
198
|
+
{"id": s.step_id, "tool": s.tool or "logic", "output": s.output}
|
|
199
|
+
for s in response.trace
|
|
200
|
+
],
|
|
201
|
+
}
|
|
202
|
+
for verification_idx in range(self.max_tries_for_plan):
|
|
203
|
+
logger.debug(
|
|
204
|
+
f"Verification planning {verification_idx + 1} of {self.max_tries_for_plan}"
|
|
205
|
+
)
|
|
206
|
+
run_trace[f"Attempt-{attempt_idx + 1}"][
|
|
207
|
+
f"Verification-{verification_idx + 1}"
|
|
208
|
+
] = datetime.now()
|
|
209
|
+
|
|
210
|
+
agent_verifier_input = AgentInput(
|
|
211
|
+
goal=goal_verifier_prompt, payload=payload_verifier
|
|
212
|
+
)
|
|
213
|
+
response_validation, feedback_, _, validation_plan = self.execute_plan(
|
|
214
|
+
agent_verifier_input, feedback_, include_state_modifying=False
|
|
215
|
+
)
|
|
216
|
+
|
|
217
|
+
if response_validation:
|
|
218
|
+
response.validated, validation_feedback = self.validated(
|
|
219
|
+
response_validation
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
if response.validated:
|
|
223
|
+
break
|
|
224
|
+
else:
|
|
225
|
+
feedback_ = {"Validation feedback": validation_feedback}
|
|
226
|
+
continue
|
|
227
|
+
else:
|
|
228
|
+
continue
|
|
229
|
+
|
|
230
|
+
if response_validation is None:
|
|
231
|
+
logger.warning(
|
|
232
|
+
f"Agent failed to verificate after {self.max_tries_for_plan} attempts"
|
|
233
|
+
)
|
|
234
|
+
continue
|
|
235
|
+
|
|
236
|
+
if response.validated:
|
|
237
|
+
break
|
|
238
|
+
|
|
239
|
+
self.rollback()
|
|
240
|
+
|
|
241
|
+
self.history.extend(
|
|
242
|
+
[
|
|
243
|
+
{
|
|
244
|
+
"role": "user",
|
|
245
|
+
"content": agent_input.goal,
|
|
246
|
+
},
|
|
247
|
+
{
|
|
248
|
+
"role": "agent",
|
|
249
|
+
"content": response.result,
|
|
250
|
+
},
|
|
251
|
+
]
|
|
252
|
+
)
|
|
253
|
+
|
|
254
|
+
self.trace.append(
|
|
255
|
+
{
|
|
256
|
+
"execution_summary": run_trace,
|
|
257
|
+
"solution_trace": response.trace,
|
|
258
|
+
"verification_trace": (
|
|
259
|
+
response_validation.trace if response_validation else ""
|
|
260
|
+
),
|
|
261
|
+
}
|
|
262
|
+
)
|
|
263
|
+
|
|
264
|
+
self.planner.goal = agent_input.goal
|
|
265
|
+
return response
|
|
266
|
+
|
|
267
|
+
def send(self, msg):
|
|
268
|
+
user_input = AgentInput(
|
|
269
|
+
goal=msg,
|
|
270
|
+
payload={
|
|
271
|
+
"chat_history": self.history,
|
|
272
|
+
},
|
|
273
|
+
)
|
|
274
|
+
response = self.run(user_input)
|
|
275
|
+
|
|
276
|
+
if isinstance(response.result, dict):
|
|
277
|
+
response.result = self.humanize_result(response)
|
|
278
|
+
else:
|
|
279
|
+
response.result = self.improve_result(response)
|
|
280
|
+
return response
|
|
281
|
+
|
|
282
|
+
def pretty_print(self, out: str):
|
|
283
|
+
self.console.print(Markdown(out))
|
|
284
|
+
|
|
285
|
+
def clear_history(self):
|
|
286
|
+
self.history = []
|
|
287
|
+
self.trace = []
|
|
288
|
+
|
|
289
|
+
def rollback(self):
|
|
290
|
+
logger.debug(f"Rolling back agent")
|
|
291
|
+
if self.rollback_function:
|
|
292
|
+
self.rollback_function()
|
|
293
|
+
else:
|
|
294
|
+
logger.debug(f"No rollback function defined")
|
|
295
|
+
|
|
296
|
+
def humanize_result(self, output: AgentOutput) -> str:
|
|
297
|
+
trace_str = ""
|
|
298
|
+
for step in output.trace:
|
|
299
|
+
trace_str += f"- {step.tool or 'logic'}: {step.output}\n"
|
|
300
|
+
|
|
301
|
+
humanize_prompt = HUMANIZER_PROMPT.format(
|
|
302
|
+
goal=self.planner.goal,
|
|
303
|
+
trace=trace_str,
|
|
304
|
+
result=output.result,
|
|
305
|
+
)
|
|
306
|
+
|
|
307
|
+
result_humanized = self.driver.generate_text(humanize_prompt)
|
|
308
|
+
logger.debug(f"Humanized response: {result_humanized}")
|
|
309
|
+
return result_humanized
|
|
310
|
+
|
|
311
|
+
def improve_result(self, response: str) -> str:
|
|
312
|
+
improve_prompt = RESPONSE_IMPROVEMENT_PROMPT.format(
|
|
313
|
+
format_response=self.planner.format_response,
|
|
314
|
+
response=response.text,
|
|
315
|
+
)
|
|
316
|
+
result_improved = self.driver.generate_text(improve_prompt)
|
|
317
|
+
logger.debug(f"Improved response: {result_improved}")
|
|
318
|
+
return result_improved
|
|
319
|
+
|
|
320
|
+
def explain_error(self, error: str):
|
|
321
|
+
error_prompt = ERROR_EXPLAIN_PROMPT.format(
|
|
322
|
+
error=error, tools=self.registry.describe()
|
|
323
|
+
)
|
|
324
|
+
error_explained = self.driver.generate_text(error_prompt)
|
|
325
|
+
return error_explained
|
|
326
|
+
|
|
327
|
+
def validated(self, response_validation: AgentOutput) -> bool:
|
|
328
|
+
evidence_str = ""
|
|
329
|
+
feedback = ""
|
|
330
|
+
|
|
331
|
+
for step in response_validation.trace:
|
|
332
|
+
evidence_str += f"- Tool: {step.tool}, Output: {step.output}\n"
|
|
333
|
+
verifier_judge_prompt = VERIFIER_JUDGE_PROMPT.format(
|
|
334
|
+
goal=self.planner.goal, evidence=evidence_str
|
|
335
|
+
)
|
|
336
|
+
|
|
337
|
+
response_text = self.driver.generate_text(verifier_judge_prompt)
|
|
338
|
+
json_match = re.search(r"\{.*\}", response_text, re.DOTALL)
|
|
339
|
+
data = json.loads(json_match.group(0))
|
|
340
|
+
|
|
341
|
+
is_valid = data.get("is_success", False)
|
|
342
|
+
|
|
343
|
+
if not is_valid:
|
|
344
|
+
logger.warning(f"Validation failed: {data.get('feedback')}")
|
|
345
|
+
feedback = data.get("feedback")
|
|
346
|
+
else:
|
|
347
|
+
logger.info(f"Attempt validated successfully")
|
|
348
|
+
|
|
349
|
+
return is_valid, feedback
|
|
350
|
+
|
|
351
|
+
def generate_chat_log(
|
|
352
|
+
self,
|
|
353
|
+
chain_messages: list[str],
|
|
354
|
+
print_chat: bool = False,
|
|
355
|
+
save_log: str = None,
|
|
356
|
+
):
|
|
357
|
+
file_handler = None
|
|
358
|
+
if save_log:
|
|
359
|
+
formatter = logging.Formatter("%(asctime)s [%(levelname)s] %(message)s")
|
|
360
|
+
file_handler = logging.FileHandler(save_log)
|
|
361
|
+
file_handler.setFormatter(formatter)
|
|
362
|
+
logger.addHandler(file_handler)
|
|
363
|
+
|
|
364
|
+
try:
|
|
365
|
+
for i, message in enumerate(chain_messages, start=1):
|
|
366
|
+
logger.debug(f"[REQUEST {i}]: '{message}'")
|
|
367
|
+
if print_chat:
|
|
368
|
+
print(f">>> {message}")
|
|
369
|
+
response = self.send(message)
|
|
370
|
+
logger.debug(f"[RESPONSE {i}]: '{response.text}'")
|
|
371
|
+
if print_chat:
|
|
372
|
+
print(f" {response.text}")
|
|
373
|
+
finally:
|
|
374
|
+
if file_handler:
|
|
375
|
+
logger.removeHandler(file_handler)
|
|
376
|
+
file_handler.close()
|
|
377
|
+
|
|
378
|
+
def save_trace(self, filename: str):
|
|
379
|
+
obj = {
|
|
380
|
+
"tools": list(self.registry.list_tools()),
|
|
381
|
+
"trace": self.trace,
|
|
382
|
+
"history": self.history,
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
if not filename.endswith(".trace"):
|
|
386
|
+
filename = f"{filename}.trace"
|
|
387
|
+
|
|
388
|
+
with open(filename, "wb") as f:
|
|
389
|
+
pickle.dump(obj, f)
|
|
390
|
+
|
|
391
|
+
def show_welcome(self):
|
|
392
|
+
self.console.clear()
|
|
393
|
+
self.pretty_print(WELCOME_MESSAGE)
|
|
394
|
+
self.pretty_print(
|
|
395
|
+
f">**Agent Profile** \n"
|
|
396
|
+
f"*{self.description}* \n"
|
|
397
|
+
f"root: {self.registry.jail_path} \n"
|
|
398
|
+
f"{self.driver}: {self.driver.model_name} \n"
|
|
399
|
+
)
|
|
400
|
+
print("\n")
|
|
401
|
+
|
|
402
|
+
def repl(self):
|
|
403
|
+
logger.setLevel(logging.CRITICAL)
|
|
404
|
+
self.show_welcome()
|
|
405
|
+
|
|
406
|
+
while True:
|
|
407
|
+
msg = prompt(
|
|
408
|
+
HTML("<ansicyan><b>>>> </b></ansicyan>"),
|
|
409
|
+
history=prompt_history,
|
|
410
|
+
)
|
|
411
|
+
|
|
412
|
+
self.console.status("[cyan]Planning...", spinner="dots")
|
|
413
|
+
|
|
414
|
+
match msg.strip():
|
|
415
|
+
case "exit;":
|
|
416
|
+
sys.exit(0)
|
|
417
|
+
|
|
418
|
+
case "clear;":
|
|
419
|
+
self.clear_history()
|
|
420
|
+
self.console.clear()
|
|
421
|
+
self.pretty_print(f">**Agent Profile** \n*{self.description}*")
|
|
422
|
+
print("\n")
|
|
423
|
+
|
|
424
|
+
case "tools;":
|
|
425
|
+
self.pretty_print(
|
|
426
|
+
self.registry.describe(
|
|
427
|
+
include_state_modifying=True, markdown=True
|
|
428
|
+
)
|
|
429
|
+
)
|
|
430
|
+
print("\n")
|
|
431
|
+
|
|
432
|
+
case "rollback;":
|
|
433
|
+
self.rollback()
|
|
434
|
+
self.pretty_print(
|
|
435
|
+
"**Rollback executed.** System reverted to the last stable state."
|
|
436
|
+
)
|
|
437
|
+
print("\n")
|
|
438
|
+
|
|
439
|
+
case _:
|
|
440
|
+
if not msg or msg.endswith(";"):
|
|
441
|
+
continue
|
|
442
|
+
|
|
443
|
+
with self.console.status(
|
|
444
|
+
"[dim]processing request[/]",
|
|
445
|
+
spinner="point",
|
|
446
|
+
spinner_style="dim",
|
|
447
|
+
):
|
|
448
|
+
output = self.send(msg)
|
|
449
|
+
self.pretty_print(f" {output.text}")
|
|
450
|
+
print("\n")
|
|
451
|
+
|
|
452
|
+
def iterate_debug(self, chain_messages: list[str], repetitions: int, path: str):
|
|
453
|
+
logger.setLevel(logging.DEBUG)
|
|
454
|
+
for i in range(repetitions):
|
|
455
|
+
self.clear_history()
|
|
456
|
+
self.rollback()
|
|
457
|
+
name = f"{self.driver.model_name}-{datetime.now().timestamp()}"
|
|
458
|
+
self.generate_chat_log(
|
|
459
|
+
chain_messages,
|
|
460
|
+
print_chat=True,
|
|
461
|
+
save_log=f"{path}/{name}.log",
|
|
462
|
+
)
|
|
463
|
+
self.save_trace(f"{path}/{name}.trace")
|
deer/core/ui.py
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
WELCOME_MESSAGE = """
|
|
2
|
+
# DEER - Deterministic Executable Engine for Runtime Agents
|
|
3
|
+
|
|
4
|
+
An LLM agent orchestration framework built for:
|
|
5
|
+
- Strict runtime control
|
|
6
|
+
- Enforced tool contracts
|
|
7
|
+
- Full execution traceability
|
|
8
|
+
- Guaranteed reproducibility
|
|
9
|
+
|
|
10
|
+
## Runtime Pipeline
|
|
11
|
+
|
|
12
|
+
1. Plan → Generate a structured execution strategy
|
|
13
|
+
2. Validate → Verify contracts, dependencies, and safety
|
|
14
|
+
3. Execute → Run approved tools and workflows
|
|
15
|
+
4. Verify → Validate outputs against expected objectives
|
|
16
|
+
|
|
17
|
+
## System Guarantees
|
|
18
|
+
|
|
19
|
+
- Deterministic execution flow
|
|
20
|
+
- Typed tool interfaces
|
|
21
|
+
- Structured validation layers
|
|
22
|
+
- Full execution traceability
|
|
23
|
+
- Backend-governed orchestration
|
|
24
|
+
|
|
25
|
+
## Commands
|
|
26
|
+
|
|
27
|
+
- `tools;` → Show tools and capabilities
|
|
28
|
+
- `clear;` → Clear the console and reset the session
|
|
29
|
+
- `exit;` → Terminate the session
|
|
30
|
+
- `rollback;` → Revert the system to the last stable state
|
|
31
|
+
"""
|
deer/drivers/__init__.py
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
import sys
|
|
3
|
+
import os
|
|
4
|
+
|
|
5
|
+
from .base_driver import LLMDriver
|
|
6
|
+
from .gemini_driver import GeminiDriver
|
|
7
|
+
from .ollama_driver import OllamaDriver
|
|
8
|
+
|
|
9
|
+
from deer.utils.console import error
|
|
10
|
+
|
|
11
|
+
backends = {
|
|
12
|
+
"gemini",
|
|
13
|
+
"openai",
|
|
14
|
+
"ollama",
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
drivers_parser = argparse.ArgumentParser(description="DEER Agent Framework CLI")
|
|
18
|
+
|
|
19
|
+
drivers_parser.add_argument(
|
|
20
|
+
"agent",
|
|
21
|
+
nargs="?",
|
|
22
|
+
help="Name of the agent to execute",
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
drivers_parser.add_argument(
|
|
26
|
+
"--backend",
|
|
27
|
+
default=os.environ.get("DEER_BACKEND"),
|
|
28
|
+
required=os.environ.get("DEER_BACKEND") is None,
|
|
29
|
+
choices=backends,
|
|
30
|
+
help="Inference backend to use",
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
drivers_parser.add_argument(
|
|
34
|
+
"--model",
|
|
35
|
+
default=os.environ.get("DEER_BACKEND_MODEL"),
|
|
36
|
+
required=os.environ.get("DEER_BACKEND_MODEL") is None,
|
|
37
|
+
help="Model identifier for the selected backend",
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def get_driver_from_parser():
|
|
42
|
+
|
|
43
|
+
args = drivers_parser.parse_args()
|
|
44
|
+
|
|
45
|
+
if args.backend not in backends:
|
|
46
|
+
error(
|
|
47
|
+
f"Unsupported backend '{args.backend}'. "
|
|
48
|
+
f"Supported backends are: {', '.join(backends)}."
|
|
49
|
+
)
|
|
50
|
+
sys.exit(1)
|
|
51
|
+
|
|
52
|
+
if not args.model:
|
|
53
|
+
error("A model identifier must be provided.")
|
|
54
|
+
sys.exit(1)
|
|
55
|
+
|
|
56
|
+
match args.backend:
|
|
57
|
+
|
|
58
|
+
case "gemini":
|
|
59
|
+
return GeminiDriver(model_name=args.model)
|
|
60
|
+
|
|
61
|
+
case "ollama":
|
|
62
|
+
return OllamaDriver(model_name=args.model)
|
|
63
|
+
|
|
64
|
+
case "openai":
|
|
65
|
+
pass
|
|
66
|
+
# return OpenAIDriver(model_name=args.model)
|