evalrun 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agents/__init__.py +6 -0
- agents/auditor/__init__.py +13 -0
- agents/auditor/budget_auditor.py +91 -0
- agents/auditor/parser.py +139 -0
- agents/auditor/prompts.py +64 -0
- agents/auditor/schema.py +67 -0
- agents/base.py +20 -0
- agents/reflection/__init__.py +3 -0
- agents/reflection/agent.py +77 -0
- agents/reflection/prompts.py +18 -0
- agents/research/__init__.py +4 -0
- agents/research/agent.py +45 -0
- agents/research/planner.py +52 -0
- agents/research/prompts.py +14 -0
- agents/support/__init__.py +5 -0
- agents/support/triage_agent.py +45 -0
- agents/travel/__init__.py +11 -0
- agents/travel/agent.py +377 -0
- agents/travel/prompts.py +30 -0
- agents/travel/session.py +110 -0
- cli/__init__.py +6 -0
- cli/demo.py +47 -0
- cli/formatter.py +93 -0
- cli/html_reporter.py +647 -0
- cli/main.py +423 -0
- cli/progress.py +38 -0
- cli/resolver.py +99 -0
- evalrun-0.4.0.dist-info/METADATA +268 -0
- evalrun-0.4.0.dist-info/RECORD +100 -0
- evalrun-0.4.0.dist-info/WHEEL +5 -0
- evalrun-0.4.0.dist-info/entry_points.txt +2 -0
- evalrun-0.4.0.dist-info/licenses/LICENSE +0 -0
- evalrun-0.4.0.dist-info/top_level.txt +4 -0
- framework/__init__.py +70 -0
- framework/core/__init__.py +17 -0
- framework/core/adapters.py +118 -0
- framework/core/contracts.py +88 -0
- framework/core/suite.py +44 -0
- framework/evaluation/__init__.py +22 -0
- framework/evaluation/base.py +29 -0
- framework/evaluation/dimensions.py +7 -0
- framework/evaluation/engine.py +110 -0
- framework/evaluation/evaluators/__init__.py +7 -0
- framework/evaluation/evaluators/adaptability.py +27 -0
- framework/evaluation/evaluators/base_llm.py +104 -0
- framework/evaluation/evaluators/constraint.py +27 -0
- framework/evaluation/evaluators/information_accuracy.py +41 -0
- framework/evaluation/evaluators/personalization.py +27 -0
- framework/evaluation/evaluators/planning.py +27 -0
- framework/evaluation/evaluators/support.py +81 -0
- framework/evaluation/prompts/__init__.py +11 -0
- framework/evaluation/prompts/adaptability.py +57 -0
- framework/evaluation/prompts/base.py +52 -0
- framework/evaluation/prompts/constraint.py +41 -0
- framework/evaluation/prompts/information_accuracy.py +79 -0
- framework/evaluation/prompts/personalization.py +57 -0
- framework/evaluation/prompts/planning.py +61 -0
- framework/evaluation/runner.py +363 -0
- framework/evaluation/testing.py +25 -0
- framework/exceptions.py +49 -0
- framework/llms/__init__.py +8 -0
- framework/llms/base.py +37 -0
- framework/llms/factory.py +38 -0
- framework/llms/gemini.py +85 -0
- framework/llms/mock.py +25 -0
- framework/llms/openai.py +94 -0
- framework/llms/openai_compatible.py +139 -0
- framework/mcp/__init__.py +20 -0
- framework/mcp/client.py +62 -0
- framework/mcp/constraints.py +125 -0
- framework/mcp/revision_summary.py +122 -0
- framework/mcp/server.py +49 -0
- framework/memory/__init__.py +3 -0
- framework/memory/base.py +17 -0
- framework/models.py +83 -0
- framework/parser.py +45 -0
- framework/parsers/__init__.py +12 -0
- framework/parsers/frontmatter.py +28 -0
- framework/parsers/mapper.py +66 -0
- framework/parsers/markdown.py +122 -0
- framework/parsers/transformers.py +112 -0
- framework/profiles/__init__.py +28 -0
- framework/profiles/registry.py +104 -0
- framework/profiles/support.py +27 -0
- framework/profiles/travel.py +78 -0
- framework/regression/__init__.py +17 -0
- framework/regression/comparator.py +273 -0
- framework/regression/loader.py +145 -0
- framework/sdk.py +151 -0
- framework/utils.py +41 -0
- framework/verification/__init__.py +13 -0
- framework/verification/base.py +32 -0
- framework/verification/extractor.py +105 -0
- framework/verification/local.py +122 -0
- framework/verification/models.py +112 -0
- framework/verification/pipeline.py +36 -0
- framework/verification/prompts.py +24 -0
- framework/verification/utils.py +54 -0
- ui/__init__.py +1 -0
- ui/server.py +255 -0
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Customer-support triage agent used by the second-domain benchmark."""
|
|
2
|
+
|
|
3
|
+
from langfuse import observe
|
|
4
|
+
|
|
5
|
+
from agents.base import BaseAgent
|
|
6
|
+
from framework.llms import BaseLLM, Message
|
|
7
|
+
from framework.models import AgentOutput
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
SUPPORT_TRIAGE_SYSTEM_PROMPT = """You are a production customer-support triage agent.
|
|
11
|
+
|
|
12
|
+
Read the ticket carefully and return a concise, structured triage decision. Include:
|
|
13
|
+
1. priority and the SLA deadline;
|
|
14
|
+
2. issue category and a one-sentence evidence-based summary;
|
|
15
|
+
3. immediate containment or next action;
|
|
16
|
+
4. the exact escalation destination and timing;
|
|
17
|
+
5. a safe, empathetic customer-facing response.
|
|
18
|
+
|
|
19
|
+
Do not invent a root cause, promise an unsupported resolution time, expose secrets,
|
|
20
|
+
or downgrade a high-impact production incident. Separate facts from hypotheses.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class SupportTriageAgent(BaseAgent):
|
|
25
|
+
"""Thin adapter that gives any BaseLLM the support-triage agent contract."""
|
|
26
|
+
|
|
27
|
+
def __init__(self, llm: BaseLLM):
|
|
28
|
+
self.llm = llm
|
|
29
|
+
|
|
30
|
+
@observe(name="support-triage-agent")
|
|
31
|
+
def run(self, prompt: str) -> AgentOutput:
|
|
32
|
+
response = self.llm.generate(
|
|
33
|
+
[
|
|
34
|
+
Message(role="system", content=SUPPORT_TRIAGE_SYSTEM_PROMPT),
|
|
35
|
+
Message(role="user", content=prompt),
|
|
36
|
+
]
|
|
37
|
+
)
|
|
38
|
+
return AgentOutput(
|
|
39
|
+
content=response.text,
|
|
40
|
+
metadata={
|
|
41
|
+
"agent": self.__class__.__name__,
|
|
42
|
+
"llm": type(self.llm).__name__,
|
|
43
|
+
"domain": "support-triage",
|
|
44
|
+
},
|
|
45
|
+
)
|
agents/travel/agent.py
ADDED
|
@@ -0,0 +1,377 @@
|
|
|
1
|
+
from langfuse import observe
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
import json
|
|
5
|
+
from typing import Any, Dict, Optional
|
|
6
|
+
from framework.llms import BaseLLM, Message
|
|
7
|
+
from framework.models import AgentOutput
|
|
8
|
+
from framework.utils import parse_json_markdown
|
|
9
|
+
from framework.memory import BaseSessionMemory
|
|
10
|
+
from agents.base import BaseAgent
|
|
11
|
+
from agents.research.planner import ResearchPlanner
|
|
12
|
+
from framework.mcp.revision_summary import (
|
|
13
|
+
REVISION_SUMMARY_JSON_TEMPLATE,
|
|
14
|
+
parse_revision_summary,
|
|
15
|
+
)
|
|
16
|
+
from .prompts import CLOSED_WORLD_EVALUATION_INSTRUCTION, TRAVEL_PLANNING_SYSTEM_PROMPT
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class TravelPlanningAgent(BaseAgent):
|
|
20
|
+
"""A travel planning assistant agent powered by an LLM.
|
|
21
|
+
|
|
22
|
+
Accepts user prompts/scenarios and plans itineraries accordingly,
|
|
23
|
+
optionally collaborating with a ResearchAgent, ResearchPlanner, and SessionMemory.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
def __init__(
|
|
27
|
+
self,
|
|
28
|
+
llm: BaseLLM,
|
|
29
|
+
research_agent: Optional[BaseAgent] = None,
|
|
30
|
+
research_planner: Optional[ResearchPlanner] = None,
|
|
31
|
+
reflection_agent: Optional[BaseAgent] = None,
|
|
32
|
+
validation_client: Optional[Any] = None,
|
|
33
|
+
):
|
|
34
|
+
"""Initializes the TravelPlanningAgent.
|
|
35
|
+
|
|
36
|
+
Args:
|
|
37
|
+
llm: The LLM client wrapper to generate itineraries.
|
|
38
|
+
research_agent: An optional research subagent to query for factual details.
|
|
39
|
+
research_planner: An optional planner to determine what queries to research.
|
|
40
|
+
reflection_agent: An optional reflection subagent to critique itineraries.
|
|
41
|
+
validation_client: Optional async MCP client for deterministic
|
|
42
|
+
replanning checks. It is opt-in so baseline experiments remain
|
|
43
|
+
unchanged.
|
|
44
|
+
"""
|
|
45
|
+
self.llm = llm
|
|
46
|
+
self.research_agent = research_agent
|
|
47
|
+
self.research_planner = research_planner or ResearchPlanner(llm)
|
|
48
|
+
self.reflection_agent = reflection_agent
|
|
49
|
+
self.validation_client = validation_client
|
|
50
|
+
|
|
51
|
+
@observe(name="travel-planning-agent")
|
|
52
|
+
def run(
|
|
53
|
+
self,
|
|
54
|
+
prompt: str,
|
|
55
|
+
session_memory: Optional[BaseSessionMemory] = None,
|
|
56
|
+
validation_scenario_id: Optional[str] = None,
|
|
57
|
+
planning_mode: str = "standard",
|
|
58
|
+
) -> AgentOutput:
|
|
59
|
+
"""Generates a travel itinerary based on user preferences.
|
|
60
|
+
|
|
61
|
+
Args:
|
|
62
|
+
prompt: The user prompt describing constraints and preferences.
|
|
63
|
+
session_memory: Optional session memory to provide state context.
|
|
64
|
+
validation_scenario_id: MCP scenario identifier. When supplied
|
|
65
|
+
together with ``validation_client`` during replanning, the
|
|
66
|
+
proposed revision is checked before the final LLM revision.
|
|
67
|
+
planning_mode: ``closed_world_evaluation`` tells the planner that a
|
|
68
|
+
benchmark deliberately supplies all planning constraints. The
|
|
69
|
+
default ``standard`` mode retains the cautious clarification
|
|
70
|
+
policy for genuinely under-specified user requests.
|
|
71
|
+
|
|
72
|
+
Returns:
|
|
73
|
+
An AgentOutput containing the planned travel itinerary.
|
|
74
|
+
"""
|
|
75
|
+
research_context = ""
|
|
76
|
+
research_metadata = []
|
|
77
|
+
|
|
78
|
+
if self.research_agent:
|
|
79
|
+
# Delegate query generation to the external planner component
|
|
80
|
+
queries = self.research_planner.plan_queries(prompt)
|
|
81
|
+
if queries:
|
|
82
|
+
findings = []
|
|
83
|
+
for query in queries:
|
|
84
|
+
try:
|
|
85
|
+
research_output = self.research_agent.run(query)
|
|
86
|
+
escaped_query = query.replace('"', '\\"')
|
|
87
|
+
# Indent multi-line answers to preserve YAML block formatting
|
|
88
|
+
indented_answer = research_output.content.replace('\n', '\n ')
|
|
89
|
+
findings.append(
|
|
90
|
+
f"- query: \"{escaped_query}\"\n"
|
|
91
|
+
f" answer: |\n"
|
|
92
|
+
f" {indented_answer}"
|
|
93
|
+
)
|
|
94
|
+
research_metadata.append({
|
|
95
|
+
"query": query,
|
|
96
|
+
"researcher": research_output.metadata.get("agent"),
|
|
97
|
+
})
|
|
98
|
+
except Exception as e:
|
|
99
|
+
findings.append(f"- query: \"{query}\"\n answer: \"FAILED: {e}\"")
|
|
100
|
+
|
|
101
|
+
research_context = (
|
|
102
|
+
"### FACTUAL RESEARCH FINDINGS (YAML format):\n"
|
|
103
|
+
"Research Findings:\n"
|
|
104
|
+
+ "\n".join(findings)
|
|
105
|
+
+ "\n==================================================\n\n"
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
memory_context = ""
|
|
109
|
+
if session_memory:
|
|
110
|
+
memory_context = (
|
|
111
|
+
"### CURRENT TRAVELER SESSION STATE (YAML format):\n"
|
|
112
|
+
f"{session_memory.to_yaml()}\n"
|
|
113
|
+
"==================================================\n\n"
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
system_prompt = self._system_prompt_for(planning_mode)
|
|
117
|
+
user_content = ""
|
|
118
|
+
if memory_context:
|
|
119
|
+
user_content += memory_context
|
|
120
|
+
if research_context:
|
|
121
|
+
user_content += research_context
|
|
122
|
+
user_content += prompt
|
|
123
|
+
|
|
124
|
+
messages = [
|
|
125
|
+
Message(role="system", content=system_prompt),
|
|
126
|
+
Message(role="user", content=user_content),
|
|
127
|
+
]
|
|
128
|
+
|
|
129
|
+
# 1. Generate initial draft plan (v1)
|
|
130
|
+
response = self.llm.generate(messages)
|
|
131
|
+
final_itinerary = response.text
|
|
132
|
+
|
|
133
|
+
# 2. Invoke reflection loop if a reflection agent is set
|
|
134
|
+
reflection_critique = None
|
|
135
|
+
mcp_validation: Optional[Dict[str, Any]] = None
|
|
136
|
+
revision_triggered = False
|
|
137
|
+
if self.reflection_agent:
|
|
138
|
+
if hasattr(self.reflection_agent, "reflect"):
|
|
139
|
+
reflection_output = self.reflection_agent.reflect(prompt, final_itinerary, session_memory)
|
|
140
|
+
else:
|
|
141
|
+
reflection_output = self.reflection_agent.run(
|
|
142
|
+
f"Scenario: {prompt}\n\nDraft Itinerary:\n{final_itinerary}"
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
critique = reflection_output.content
|
|
146
|
+
|
|
147
|
+
# v2.1 must validate every replanning draft. Reflection approval
|
|
148
|
+
# does not prove that hard constraints or savings arithmetic hold.
|
|
149
|
+
is_replanning = False
|
|
150
|
+
if session_memory and session_memory.state.current_day > 1:
|
|
151
|
+
is_replanning = True
|
|
152
|
+
elif "currently on day" in prompt.lower() or "replanning" in prompt.lower() or "disruption" in prompt.lower():
|
|
153
|
+
is_replanning = True
|
|
154
|
+
|
|
155
|
+
initial_validation = None
|
|
156
|
+
if is_replanning and self.validation_client and validation_scenario_id:
|
|
157
|
+
initial_validation = self._validate_replanning_proposal(
|
|
158
|
+
system_prompt=system_prompt,
|
|
159
|
+
scenario_prompt=prompt,
|
|
160
|
+
draft_itinerary=final_itinerary,
|
|
161
|
+
critique=critique,
|
|
162
|
+
scenario_id=validation_scenario_id,
|
|
163
|
+
)
|
|
164
|
+
mcp_validation = {"initial": initial_validation, "final": None}
|
|
165
|
+
|
|
166
|
+
reflection_requires_revision = "ITINERARY APPROVED" not in critique
|
|
167
|
+
mcp_requires_revision = self._validation_requires_revision(initial_validation)
|
|
168
|
+
revision_triggered = reflection_requires_revision or mcp_requires_revision
|
|
169
|
+
|
|
170
|
+
if reflection_requires_revision:
|
|
171
|
+
reflection_critique = critique
|
|
172
|
+
|
|
173
|
+
if revision_triggered:
|
|
174
|
+
# Increment plan version in memory
|
|
175
|
+
if session_memory:
|
|
176
|
+
session_memory.state.current_plan_version += 1
|
|
177
|
+
|
|
178
|
+
# Regenerate memory context with updated plan version
|
|
179
|
+
memory_context_v2 = ""
|
|
180
|
+
if session_memory:
|
|
181
|
+
memory_context_v2 = (
|
|
182
|
+
"### CURRENT TRAVELER SESSION STATE (YAML format):\n"
|
|
183
|
+
f"{session_memory.to_yaml()}\n"
|
|
184
|
+
"==================================================\n\n"
|
|
185
|
+
)
|
|
186
|
+
|
|
187
|
+
if is_replanning:
|
|
188
|
+
revision_rules = (
|
|
189
|
+
"1. PRESERVE every part of the itinerary that was NOT criticized. Do NOT rewrite or shift unaffected days.\n"
|
|
190
|
+
"2. Do NOT modify, delete, or rearrange locked bookings (e.g., booked flights, non-refundable hotel stays).\n"
|
|
191
|
+
"3. Make ONLY the minimum necessary adjustments needed to resolve the critique points."
|
|
192
|
+
)
|
|
193
|
+
else:
|
|
194
|
+
revision_rules = (
|
|
195
|
+
"1. You are free to re-sequence, compress, or optimize the entire itinerary globally to satisfy the constraints (e.g., total duration, must-visit destinations) and address the critiques.\n"
|
|
196
|
+
"2. Do NOT exceed the total duration limit specified in the original request.\n"
|
|
197
|
+
"3. Make adjustments to address the critiques while keeping the travel style and preferences consistent."
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
# Formulate revision prompt
|
|
201
|
+
validation_context = ""
|
|
202
|
+
if initial_validation:
|
|
203
|
+
validation_context = (
|
|
204
|
+
"\n\n### DETERMINISTIC MCP VALIDATION REPORT:\n"
|
|
205
|
+
f"{json.dumps(initial_validation, indent=2)}\n"
|
|
206
|
+
"Treat every violation or unmet savings target in this report as a hard "
|
|
207
|
+
"requirement for the final revision. Do not claim a saving unless it is "
|
|
208
|
+
"represented by a concrete itinerary change.\n"
|
|
209
|
+
)
|
|
210
|
+
revision_prompt = (
|
|
211
|
+
f"### ORIGINAL SCENARIO REQUEST:\n{prompt}\n\n"
|
|
212
|
+
f"### INITIAL DRAFT PLAN:\n{final_itinerary}\n\n"
|
|
213
|
+
f"### REFLECTION CRITIQUE:\n{critique}\n\n"
|
|
214
|
+
"You are instructed to revise the initial draft plan to resolve the critiques listed above.\n"
|
|
215
|
+
"You MUST strictly follow these rules during the revision:\n"
|
|
216
|
+
f"{revision_rules}\n"
|
|
217
|
+
"4. If a critique point directly conflicts with an existing hard constraint in the original request, prioritize and preserve the hard constraint.\n"
|
|
218
|
+
"5. You MUST generate a complete, full day-by-day traveler-facing itinerary. Do NOT output a list of questions, deferrals, or requests for information in place of the itinerary."
|
|
219
|
+
f"{validation_context}"
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
user_content_v2 = ""
|
|
223
|
+
if memory_context_v2:
|
|
224
|
+
user_content_v2 += memory_context_v2
|
|
225
|
+
if research_context:
|
|
226
|
+
user_content_v2 += research_context
|
|
227
|
+
user_content_v2 += revision_prompt
|
|
228
|
+
|
|
229
|
+
messages_v2 = [
|
|
230
|
+
Message(role="system", content=system_prompt),
|
|
231
|
+
Message(role="user", content=user_content_v2),
|
|
232
|
+
]
|
|
233
|
+
# Generate revised plan (v2)
|
|
234
|
+
response = self.llm.generate(messages_v2)
|
|
235
|
+
if response.text.strip():
|
|
236
|
+
final_itinerary = response.text
|
|
237
|
+
|
|
238
|
+
if is_replanning and self.validation_client and validation_scenario_id:
|
|
239
|
+
mcp_validation["final"] = self._validate_replanning_proposal(
|
|
240
|
+
system_prompt=system_prompt,
|
|
241
|
+
scenario_prompt=prompt,
|
|
242
|
+
draft_itinerary=final_itinerary,
|
|
243
|
+
critique="Validate the final revised itinerary against the original constraints.",
|
|
244
|
+
scenario_id=validation_scenario_id,
|
|
245
|
+
)
|
|
246
|
+
|
|
247
|
+
# 3. Assemble metadata
|
|
248
|
+
metadata = {
|
|
249
|
+
"agent": self.__class__.__name__,
|
|
250
|
+
"llm": type(self.llm).__name__,
|
|
251
|
+
"reflection_approved": not bool(reflection_critique),
|
|
252
|
+
"revision_triggered": revision_triggered,
|
|
253
|
+
}
|
|
254
|
+
if research_metadata:
|
|
255
|
+
metadata["research_steps"] = research_metadata
|
|
256
|
+
if revision_triggered:
|
|
257
|
+
metadata["reflection_critique"] = reflection_critique
|
|
258
|
+
metadata["final_plan_version"] = session_memory.state.current_plan_version if session_memory else 2
|
|
259
|
+
if mcp_validation:
|
|
260
|
+
metadata["mcp_validation"] = mcp_validation
|
|
261
|
+
|
|
262
|
+
return AgentOutput(
|
|
263
|
+
content=final_itinerary,
|
|
264
|
+
metadata=metadata,
|
|
265
|
+
)
|
|
266
|
+
|
|
267
|
+
def _validate_replanning_proposal(
|
|
268
|
+
self,
|
|
269
|
+
system_prompt: str,
|
|
270
|
+
scenario_prompt: str,
|
|
271
|
+
draft_itinerary: str,
|
|
272
|
+
critique: str,
|
|
273
|
+
scenario_id: str,
|
|
274
|
+
) -> Dict[str, Any]:
|
|
275
|
+
"""Gets explicit revision claims and validates them through MCP.
|
|
276
|
+
|
|
277
|
+
The model does not use this call to write the final traveler-facing
|
|
278
|
+
answer. It only declares the concrete changes it intends to make, which
|
|
279
|
+
lets MCP verify locked bookings and arithmetic before the final rewrite.
|
|
280
|
+
A malformed summary or unavailable server is reported to the final
|
|
281
|
+
revision prompt rather than silently treated as a passing check.
|
|
282
|
+
"""
|
|
283
|
+
# Retrieve exact locked constraints to ground the model on valid booking IDs
|
|
284
|
+
locked_info = {}
|
|
285
|
+
if self.validation_client:
|
|
286
|
+
try:
|
|
287
|
+
locked_info = self._call_validation_tool(
|
|
288
|
+
"get_locked_constraints", {"scenario_id": scenario_id}
|
|
289
|
+
)
|
|
290
|
+
except Exception:
|
|
291
|
+
pass
|
|
292
|
+
|
|
293
|
+
immutable_ids = locked_info.get("immutable_booking_ids", ["kyoto-hostel", "narita-return-flight"])
|
|
294
|
+
|
|
295
|
+
summary_prompt = (
|
|
296
|
+
"Return only one JSON object describing the proposed replanning revision. "
|
|
297
|
+
"Do not write an itinerary and do not use Markdown outside the JSON.\n\n"
|
|
298
|
+
f"The scenario_id must be exactly: {scenario_id}\n"
|
|
299
|
+
f"The exact booking_id values to preserve in booking_actions MUST be: {json.dumps(immutable_ids)}. "
|
|
300
|
+
"Do NOT add suffixes or alter these booking_id strings.\n"
|
|
301
|
+
"List every itinerary day you intend to change. For every locked booking, "
|
|
302
|
+
"declare preserve, move, cancel, or modify. Itemize each concrete saving "
|
|
303
|
+
"in INR; do not estimate a total without its components.\n\n"
|
|
304
|
+
f"Required JSON shape:\n{REVISION_SUMMARY_JSON_TEMPLATE}\n\n"
|
|
305
|
+
f"### ORIGINAL SCENARIO REQUEST:\n{scenario_prompt}\n\n"
|
|
306
|
+
f"### INITIAL DRAFT PLAN:\n{draft_itinerary}\n\n"
|
|
307
|
+
f"### REFLECTION CRITIQUE:\n{critique}\n"
|
|
308
|
+
)
|
|
309
|
+
summary_response = self.llm.generate(
|
|
310
|
+
[
|
|
311
|
+
Message(role="system", content=system_prompt),
|
|
312
|
+
Message(role="user", content=summary_prompt),
|
|
313
|
+
]
|
|
314
|
+
)
|
|
315
|
+
|
|
316
|
+
try:
|
|
317
|
+
summary = parse_revision_summary(summary_response.text)
|
|
318
|
+
if summary.scenario_id != scenario_id:
|
|
319
|
+
raise ValueError(
|
|
320
|
+
"Revision summary scenario_id does not match the requested scenario."
|
|
321
|
+
)
|
|
322
|
+
tool_arguments = summary.to_tool_arguments()
|
|
323
|
+
return {
|
|
324
|
+
"status": "completed",
|
|
325
|
+
"revision_summary": tool_arguments,
|
|
326
|
+
"locked_constraints": locked_info or self._call_validation_tool(
|
|
327
|
+
"get_locked_constraints", {"scenario_id": scenario_id}
|
|
328
|
+
),
|
|
329
|
+
"revision_check": self._call_validation_tool(
|
|
330
|
+
"validate_revision",
|
|
331
|
+
{
|
|
332
|
+
"scenario_id": scenario_id,
|
|
333
|
+
"booking_actions": tool_arguments["booking_actions"],
|
|
334
|
+
"changed_days": tool_arguments["changed_days"],
|
|
335
|
+
},
|
|
336
|
+
),
|
|
337
|
+
"savings_check": self._call_validation_tool(
|
|
338
|
+
"calculate_savings",
|
|
339
|
+
{
|
|
340
|
+
"scenario_id": scenario_id,
|
|
341
|
+
"savings_items": tool_arguments["savings_items"],
|
|
342
|
+
},
|
|
343
|
+
),
|
|
344
|
+
}
|
|
345
|
+
except (RuntimeError, ValueError) as error:
|
|
346
|
+
return {
|
|
347
|
+
"status": "unavailable",
|
|
348
|
+
"error": str(error),
|
|
349
|
+
"raw_revision_summary": summary_response.text,
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
def _call_validation_tool(self, name: str, arguments: Dict[str, Any]) -> Dict[str, Any]:
|
|
353
|
+
"""Runs an async MCP call from this currently synchronous agent API."""
|
|
354
|
+
return asyncio.run(self.validation_client.call_tool(name, arguments))
|
|
355
|
+
|
|
356
|
+
@staticmethod
|
|
357
|
+
def _validation_requires_revision(validation: Optional[Dict[str, Any]]) -> bool:
|
|
358
|
+
"""Returns whether an MCP result requires a constrained revision."""
|
|
359
|
+
if validation is None:
|
|
360
|
+
return False
|
|
361
|
+
if validation.get("status") != "completed":
|
|
362
|
+
return True
|
|
363
|
+
return (
|
|
364
|
+
not validation.get("revision_check", {}).get("valid", False)
|
|
365
|
+
or not validation.get("savings_check", {}).get("target_met", False)
|
|
366
|
+
)
|
|
367
|
+
|
|
368
|
+
@staticmethod
|
|
369
|
+
def _system_prompt_for(planning_mode: str) -> str:
|
|
370
|
+
"""Returns the planner policy for normal or controlled benchmark runs."""
|
|
371
|
+
if planning_mode == "standard":
|
|
372
|
+
return TRAVEL_PLANNING_SYSTEM_PROMPT
|
|
373
|
+
if planning_mode == "closed_world_evaluation":
|
|
374
|
+
return TRAVEL_PLANNING_SYSTEM_PROMPT + "\n" + CLOSED_WORLD_EVALUATION_INSTRUCTION
|
|
375
|
+
raise ValueError(
|
|
376
|
+
"planning_mode must be 'standard' or 'closed_world_evaluation'."
|
|
377
|
+
)
|
agents/travel/prompts.py
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
TRAVEL_PLANNING_SYSTEM_PROMPT = (
|
|
2
|
+
"You are a professional travel agent assistant.\n"
|
|
3
|
+
"Your task is to plan a detailed, well-structured travel itinerary "
|
|
4
|
+
"based on the user's preferences, budget limits, duration, backpacking style, "
|
|
5
|
+
"and other specified constraints.\n"
|
|
6
|
+
"Be precise, geographically efficient, and mention exact travel logistics, closed days, and pricing "
|
|
7
|
+
"where appropriate.\n\n"
|
|
8
|
+
"CRITICAL RULE ON UNCERTAINTY:\n"
|
|
9
|
+
"If the user's request lacks critical details needed to build a realistic itinerary (such as travel dates, "
|
|
10
|
+
"trip duration, specific destinations, or seasonal timing dependencies like cherry blossom bloom forecasts), "
|
|
11
|
+
"you must NOT guess, assume, or hallucinate a sample or final itinerary. Instead, you must:\n"
|
|
12
|
+
"1. Defer generating the final day-by-day itinerary.\n"
|
|
13
|
+
"2. Ask the user clarifying questions to obtain the missing details (e.g., travel style, departure airport).\n"
|
|
14
|
+
"3. Output structured tool requests in a YAML block specifying what external search queries are required to gather "
|
|
15
|
+
"the necessary data first (forecasts, festival dates, pricing). Use this exact format:\n\n"
|
|
16
|
+
"Tool Requests:\n"
|
|
17
|
+
" - Search:\n"
|
|
18
|
+
" query: \"<search query>\"\n"
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
CLOSED_WORLD_EVALUATION_INSTRUCTION = (
|
|
23
|
+
"EVALUATION SCENARIO MODE:\n"
|
|
24
|
+
"The scenario supplied in the user message is a complete, closed-world test "
|
|
25
|
+
"case. Produce a concrete provisional itinerary using its stated days, "
|
|
26
|
+
"constraints, and disruptions. Do not ask clarifying questions or defer the "
|
|
27
|
+
"itinerary merely because calendar dates, live availability, or exact prices "
|
|
28
|
+
"are absent. Do not invent external facts: label any operational detail that "
|
|
29
|
+
"would require confirmation as conditional, while still completing the plan.\n"
|
|
30
|
+
)
|
agents/travel/session.py
ADDED
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
"""Travel-specific session memory implementation."""
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field, asdict
|
|
4
|
+
from typing import Any, Dict, List, Optional
|
|
5
|
+
import yaml
|
|
6
|
+
import json
|
|
7
|
+
from framework.memory import BaseSessionMemory
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass
|
|
11
|
+
class Booking:
|
|
12
|
+
"""Represents a structured locked booking."""
|
|
13
|
+
booking_id: str
|
|
14
|
+
booking_type: str # e.g., "flight", "hotel", "activity", "train"
|
|
15
|
+
location: str
|
|
16
|
+
start_day: int
|
|
17
|
+
end_day: int
|
|
18
|
+
refundable: bool = True
|
|
19
|
+
notes: str = ""
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class RemoteWorkSchedule:
|
|
24
|
+
"""Represents a remote work commitment and timezone constraints."""
|
|
25
|
+
timezone: str = "UTC"
|
|
26
|
+
availability_start: str = "09:00"
|
|
27
|
+
availability_end: str = "17:00"
|
|
28
|
+
meeting_start: str = ""
|
|
29
|
+
meeting_end: str = ""
|
|
30
|
+
flexible_hours: int = 0
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass
|
|
34
|
+
class UserPreferences:
|
|
35
|
+
"""Represents purely static user profiling and travel preferences."""
|
|
36
|
+
interests: List[str] = field(default_factory=list)
|
|
37
|
+
travel_style: str = "balanced"
|
|
38
|
+
accommodation_preference: str = "hotel"
|
|
39
|
+
walking_tolerance: str = "moderate"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class TripConstraints:
|
|
44
|
+
"""Represents hard trip limits, schedules, and locked bookings."""
|
|
45
|
+
total_budget: float = 0.0
|
|
46
|
+
duration_days: int = 1
|
|
47
|
+
destinations: List[str] = field(default_factory=list)
|
|
48
|
+
remote_work_schedule: Optional[RemoteWorkSchedule] = None
|
|
49
|
+
locked_bookings: List[Booking] = field(default_factory=list)
|
|
50
|
+
hard_constraints: List[str] = field(default_factory=list)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass
|
|
54
|
+
class CurrentTripState:
|
|
55
|
+
"""Tracks active planning state and execution progress."""
|
|
56
|
+
current_day: int = 1
|
|
57
|
+
remaining_budget: float = 0.0
|
|
58
|
+
completed_destinations: List[str] = field(default_factory=list)
|
|
59
|
+
current_plan_version: int = 1
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class TravelSessionMemory(BaseSessionMemory):
|
|
63
|
+
"""Holds and serializes working session memory for travel planning and reflection."""
|
|
64
|
+
|
|
65
|
+
def __init__(
|
|
66
|
+
self,
|
|
67
|
+
session_id: str,
|
|
68
|
+
version: int = 1,
|
|
69
|
+
preferences: Optional[UserPreferences] = None,
|
|
70
|
+
constraints: Optional[TripConstraints] = None,
|
|
71
|
+
state: Optional[CurrentTripState] = None,
|
|
72
|
+
):
|
|
73
|
+
"""Initializes the TravelSessionMemory.
|
|
74
|
+
|
|
75
|
+
Args:
|
|
76
|
+
session_id: A unique string identifier for the planning session.
|
|
77
|
+
version: Schema/memory state version.
|
|
78
|
+
preferences: Optional initial UserPreferences.
|
|
79
|
+
constraints: Optional initial TripConstraints.
|
|
80
|
+
state: Optional initial CurrentTripState.
|
|
81
|
+
"""
|
|
82
|
+
self.session_id = session_id
|
|
83
|
+
self.version = version
|
|
84
|
+
self.preferences = preferences or UserPreferences()
|
|
85
|
+
self.constraints = constraints or TripConstraints()
|
|
86
|
+
self.state = state or CurrentTripState()
|
|
87
|
+
|
|
88
|
+
def to_yaml(self) -> str:
|
|
89
|
+
"""Serializes session state to structured YAML for LLM context injection.
|
|
90
|
+
|
|
91
|
+
Excludes empty lists or empty dicts to conserve context tokens.
|
|
92
|
+
"""
|
|
93
|
+
data = {
|
|
94
|
+
"session_id": self.session_id,
|
|
95
|
+
"version": self.version,
|
|
96
|
+
"preferences": {k: v for k, v in asdict(self.preferences).items() if v},
|
|
97
|
+
"constraints": {k: v for k, v in asdict(self.constraints).items() if v},
|
|
98
|
+
"state": {k: v for k, v in asdict(self.state).items() if v},
|
|
99
|
+
}
|
|
100
|
+
return yaml.dump(data, sort_keys=False, default_flow_style=False)
|
|
101
|
+
|
|
102
|
+
def to_json(self) -> str:
|
|
103
|
+
"""Returns JSON representation of memory state."""
|
|
104
|
+
return json.dumps({
|
|
105
|
+
"session_id": self.session_id,
|
|
106
|
+
"version": self.version,
|
|
107
|
+
"preferences": asdict(self.preferences),
|
|
108
|
+
"constraints": asdict(self.constraints),
|
|
109
|
+
"state": asdict(self.state),
|
|
110
|
+
}, indent=2)
|
cli/__init__.py
ADDED
cli/demo.py
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""Offline, zero-credential EvalRun product demonstration."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from datetime import datetime, timezone
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from cli.formatter import format_terminal_summary
|
|
8
|
+
from cli.html_reporter import generate_html_report
|
|
9
|
+
from framework.models import DimensionScore, EvaluationResult
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def run_demo(output_dir: str = "results/demo") -> int:
|
|
13
|
+
"""Create a deterministic sample report without contacting a model endpoint."""
|
|
14
|
+
out = Path(output_dir)
|
|
15
|
+
out.mkdir(parents=True, exist_ok=True)
|
|
16
|
+
result = EvaluationResult(
|
|
17
|
+
benchmark_id="evalrun-demo-scenario",
|
|
18
|
+
benchmark_name="EvalRun Offline Demo",
|
|
19
|
+
overall_score=88.0,
|
|
20
|
+
dimension_scores=[
|
|
21
|
+
DimensionScore("Constraint Satisfaction", 90.0, "All demo constraints were satisfied."),
|
|
22
|
+
DimensionScore("Planning Quality", 86.0, "The demo output is coherent and complete."),
|
|
23
|
+
],
|
|
24
|
+
passed=True,
|
|
25
|
+
agent_metadata={"demo": True, "audit_gate_decision": "PASS"},
|
|
26
|
+
)
|
|
27
|
+
manifest = {
|
|
28
|
+
"run_id": "evalrun-offline-demo",
|
|
29
|
+
"timestamp_utc": datetime.now(timezone.utc).isoformat(),
|
|
30
|
+
"target_agent_spec": "built-in offline demo",
|
|
31
|
+
"target_model": {"model_name": "offline-demo", "base_url": "local", "api_key": "EMPTY"},
|
|
32
|
+
"judge_model": {"model_name": "offline-demo", "base_url": "local", "api_key": "EMPTY"},
|
|
33
|
+
"output_dir": str(out),
|
|
34
|
+
"total_scenarios": 1,
|
|
35
|
+
"overall_passed": True,
|
|
36
|
+
"scenarios": [{"scenario_id": result.benchmark_id, "scenario_name": result.benchmark_name,
|
|
37
|
+
"overall_score": result.overall_score, "passed": result.passed,
|
|
38
|
+
"audit_gate_decision": "PASS"}],
|
|
39
|
+
}
|
|
40
|
+
with open(out / "manifest.json", "w", encoding="utf-8") as handle:
|
|
41
|
+
json.dump(manifest, handle, indent=2)
|
|
42
|
+
with open(out / "evalrun-demo-output.txt", "w", encoding="utf-8") as handle:
|
|
43
|
+
handle.write("This is a deterministic offline preview. No model endpoint was contacted.\n")
|
|
44
|
+
generate_html_report([result], manifest, str(out))
|
|
45
|
+
print(format_terminal_summary([result], manifest))
|
|
46
|
+
print(f"\nOffline demo artifacts: {out.resolve()}")
|
|
47
|
+
return 0
|