evalrun 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. agents/__init__.py +6 -0
  2. agents/auditor/__init__.py +13 -0
  3. agents/auditor/budget_auditor.py +91 -0
  4. agents/auditor/parser.py +139 -0
  5. agents/auditor/prompts.py +64 -0
  6. agents/auditor/schema.py +67 -0
  7. agents/base.py +20 -0
  8. agents/reflection/__init__.py +3 -0
  9. agents/reflection/agent.py +77 -0
  10. agents/reflection/prompts.py +18 -0
  11. agents/research/__init__.py +4 -0
  12. agents/research/agent.py +45 -0
  13. agents/research/planner.py +52 -0
  14. agents/research/prompts.py +14 -0
  15. agents/support/__init__.py +5 -0
  16. agents/support/triage_agent.py +45 -0
  17. agents/travel/__init__.py +11 -0
  18. agents/travel/agent.py +377 -0
  19. agents/travel/prompts.py +30 -0
  20. agents/travel/session.py +110 -0
  21. cli/__init__.py +6 -0
  22. cli/demo.py +47 -0
  23. cli/formatter.py +93 -0
  24. cli/html_reporter.py +647 -0
  25. cli/main.py +423 -0
  26. cli/progress.py +38 -0
  27. cli/resolver.py +99 -0
  28. evalrun-0.4.0.dist-info/METADATA +268 -0
  29. evalrun-0.4.0.dist-info/RECORD +100 -0
  30. evalrun-0.4.0.dist-info/WHEEL +5 -0
  31. evalrun-0.4.0.dist-info/entry_points.txt +2 -0
  32. evalrun-0.4.0.dist-info/licenses/LICENSE +0 -0
  33. evalrun-0.4.0.dist-info/top_level.txt +4 -0
  34. framework/__init__.py +70 -0
  35. framework/core/__init__.py +17 -0
  36. framework/core/adapters.py +118 -0
  37. framework/core/contracts.py +88 -0
  38. framework/core/suite.py +44 -0
  39. framework/evaluation/__init__.py +22 -0
  40. framework/evaluation/base.py +29 -0
  41. framework/evaluation/dimensions.py +7 -0
  42. framework/evaluation/engine.py +110 -0
  43. framework/evaluation/evaluators/__init__.py +7 -0
  44. framework/evaluation/evaluators/adaptability.py +27 -0
  45. framework/evaluation/evaluators/base_llm.py +104 -0
  46. framework/evaluation/evaluators/constraint.py +27 -0
  47. framework/evaluation/evaluators/information_accuracy.py +41 -0
  48. framework/evaluation/evaluators/personalization.py +27 -0
  49. framework/evaluation/evaluators/planning.py +27 -0
  50. framework/evaluation/evaluators/support.py +81 -0
  51. framework/evaluation/prompts/__init__.py +11 -0
  52. framework/evaluation/prompts/adaptability.py +57 -0
  53. framework/evaluation/prompts/base.py +52 -0
  54. framework/evaluation/prompts/constraint.py +41 -0
  55. framework/evaluation/prompts/information_accuracy.py +79 -0
  56. framework/evaluation/prompts/personalization.py +57 -0
  57. framework/evaluation/prompts/planning.py +61 -0
  58. framework/evaluation/runner.py +363 -0
  59. framework/evaluation/testing.py +25 -0
  60. framework/exceptions.py +49 -0
  61. framework/llms/__init__.py +8 -0
  62. framework/llms/base.py +37 -0
  63. framework/llms/factory.py +38 -0
  64. framework/llms/gemini.py +85 -0
  65. framework/llms/mock.py +25 -0
  66. framework/llms/openai.py +94 -0
  67. framework/llms/openai_compatible.py +139 -0
  68. framework/mcp/__init__.py +20 -0
  69. framework/mcp/client.py +62 -0
  70. framework/mcp/constraints.py +125 -0
  71. framework/mcp/revision_summary.py +122 -0
  72. framework/mcp/server.py +49 -0
  73. framework/memory/__init__.py +3 -0
  74. framework/memory/base.py +17 -0
  75. framework/models.py +83 -0
  76. framework/parser.py +45 -0
  77. framework/parsers/__init__.py +12 -0
  78. framework/parsers/frontmatter.py +28 -0
  79. framework/parsers/mapper.py +66 -0
  80. framework/parsers/markdown.py +122 -0
  81. framework/parsers/transformers.py +112 -0
  82. framework/profiles/__init__.py +28 -0
  83. framework/profiles/registry.py +104 -0
  84. framework/profiles/support.py +27 -0
  85. framework/profiles/travel.py +78 -0
  86. framework/regression/__init__.py +17 -0
  87. framework/regression/comparator.py +273 -0
  88. framework/regression/loader.py +145 -0
  89. framework/sdk.py +151 -0
  90. framework/utils.py +41 -0
  91. framework/verification/__init__.py +13 -0
  92. framework/verification/base.py +32 -0
  93. framework/verification/extractor.py +105 -0
  94. framework/verification/local.py +122 -0
  95. framework/verification/models.py +112 -0
  96. framework/verification/pipeline.py +36 -0
  97. framework/verification/prompts.py +24 -0
  98. framework/verification/utils.py +54 -0
  99. ui/__init__.py +1 -0
  100. ui/server.py +255 -0
@@ -0,0 +1,45 @@
1
+ """Customer-support triage agent used by the second-domain benchmark."""
2
+
3
+ from langfuse import observe
4
+
5
+ from agents.base import BaseAgent
6
+ from framework.llms import BaseLLM, Message
7
+ from framework.models import AgentOutput
8
+
9
+
10
+ SUPPORT_TRIAGE_SYSTEM_PROMPT = """You are a production customer-support triage agent.
11
+
12
+ Read the ticket carefully and return a concise, structured triage decision. Include:
13
+ 1. priority and the SLA deadline;
14
+ 2. issue category and a one-sentence evidence-based summary;
15
+ 3. immediate containment or next action;
16
+ 4. the exact escalation destination and timing;
17
+ 5. a safe, empathetic customer-facing response.
18
+
19
+ Do not invent a root cause, promise an unsupported resolution time, expose secrets,
20
+ or downgrade a high-impact production incident. Separate facts from hypotheses.
21
+ """
22
+
23
+
24
+ class SupportTriageAgent(BaseAgent):
25
+ """Thin adapter that gives any BaseLLM the support-triage agent contract."""
26
+
27
+ def __init__(self, llm: BaseLLM):
28
+ self.llm = llm
29
+
30
+ @observe(name="support-triage-agent")
31
+ def run(self, prompt: str) -> AgentOutput:
32
+ response = self.llm.generate(
33
+ [
34
+ Message(role="system", content=SUPPORT_TRIAGE_SYSTEM_PROMPT),
35
+ Message(role="user", content=prompt),
36
+ ]
37
+ )
38
+ return AgentOutput(
39
+ content=response.text,
40
+ metadata={
41
+ "agent": self.__class__.__name__,
42
+ "llm": type(self.llm).__name__,
43
+ "domain": "support-triage",
44
+ },
45
+ )
@@ -0,0 +1,11 @@
1
+ """Travel planning agent package."""
2
+
3
+ from agents.travel.agent import TravelPlanningAgent
4
+ from agents.travel.session import (
5
+ Booking,
6
+ RemoteWorkSchedule,
7
+ UserPreferences,
8
+ TripConstraints,
9
+ CurrentTripState,
10
+ TravelSessionMemory,
11
+ )
agents/travel/agent.py ADDED
@@ -0,0 +1,377 @@
1
+ from langfuse import observe
2
+
3
+ import asyncio
4
+ import json
5
+ from typing import Any, Dict, Optional
6
+ from framework.llms import BaseLLM, Message
7
+ from framework.models import AgentOutput
8
+ from framework.utils import parse_json_markdown
9
+ from framework.memory import BaseSessionMemory
10
+ from agents.base import BaseAgent
11
+ from agents.research.planner import ResearchPlanner
12
+ from framework.mcp.revision_summary import (
13
+ REVISION_SUMMARY_JSON_TEMPLATE,
14
+ parse_revision_summary,
15
+ )
16
+ from .prompts import CLOSED_WORLD_EVALUATION_INSTRUCTION, TRAVEL_PLANNING_SYSTEM_PROMPT
17
+
18
+
19
+ class TravelPlanningAgent(BaseAgent):
20
+ """A travel planning assistant agent powered by an LLM.
21
+
22
+ Accepts user prompts/scenarios and plans itineraries accordingly,
23
+ optionally collaborating with a ResearchAgent, ResearchPlanner, and SessionMemory.
24
+ """
25
+
26
+ def __init__(
27
+ self,
28
+ llm: BaseLLM,
29
+ research_agent: Optional[BaseAgent] = None,
30
+ research_planner: Optional[ResearchPlanner] = None,
31
+ reflection_agent: Optional[BaseAgent] = None,
32
+ validation_client: Optional[Any] = None,
33
+ ):
34
+ """Initializes the TravelPlanningAgent.
35
+
36
+ Args:
37
+ llm: The LLM client wrapper to generate itineraries.
38
+ research_agent: An optional research subagent to query for factual details.
39
+ research_planner: An optional planner to determine what queries to research.
40
+ reflection_agent: An optional reflection subagent to critique itineraries.
41
+ validation_client: Optional async MCP client for deterministic
42
+ replanning checks. It is opt-in so baseline experiments remain
43
+ unchanged.
44
+ """
45
+ self.llm = llm
46
+ self.research_agent = research_agent
47
+ self.research_planner = research_planner or ResearchPlanner(llm)
48
+ self.reflection_agent = reflection_agent
49
+ self.validation_client = validation_client
50
+
51
+ @observe(name="travel-planning-agent")
52
+ def run(
53
+ self,
54
+ prompt: str,
55
+ session_memory: Optional[BaseSessionMemory] = None,
56
+ validation_scenario_id: Optional[str] = None,
57
+ planning_mode: str = "standard",
58
+ ) -> AgentOutput:
59
+ """Generates a travel itinerary based on user preferences.
60
+
61
+ Args:
62
+ prompt: The user prompt describing constraints and preferences.
63
+ session_memory: Optional session memory to provide state context.
64
+ validation_scenario_id: MCP scenario identifier. When supplied
65
+ together with ``validation_client`` during replanning, the
66
+ proposed revision is checked before the final LLM revision.
67
+ planning_mode: ``closed_world_evaluation`` tells the planner that a
68
+ benchmark deliberately supplies all planning constraints. The
69
+ default ``standard`` mode retains the cautious clarification
70
+ policy for genuinely under-specified user requests.
71
+
72
+ Returns:
73
+ An AgentOutput containing the planned travel itinerary.
74
+ """
75
+ research_context = ""
76
+ research_metadata = []
77
+
78
+ if self.research_agent:
79
+ # Delegate query generation to the external planner component
80
+ queries = self.research_planner.plan_queries(prompt)
81
+ if queries:
82
+ findings = []
83
+ for query in queries:
84
+ try:
85
+ research_output = self.research_agent.run(query)
86
+ escaped_query = query.replace('"', '\\"')
87
+ # Indent multi-line answers to preserve YAML block formatting
88
+ indented_answer = research_output.content.replace('\n', '\n ')
89
+ findings.append(
90
+ f"- query: \"{escaped_query}\"\n"
91
+ f" answer: |\n"
92
+ f" {indented_answer}"
93
+ )
94
+ research_metadata.append({
95
+ "query": query,
96
+ "researcher": research_output.metadata.get("agent"),
97
+ })
98
+ except Exception as e:
99
+ findings.append(f"- query: \"{query}\"\n answer: \"FAILED: {e}\"")
100
+
101
+ research_context = (
102
+ "### FACTUAL RESEARCH FINDINGS (YAML format):\n"
103
+ "Research Findings:\n"
104
+ + "\n".join(findings)
105
+ + "\n==================================================\n\n"
106
+ )
107
+
108
+ memory_context = ""
109
+ if session_memory:
110
+ memory_context = (
111
+ "### CURRENT TRAVELER SESSION STATE (YAML format):\n"
112
+ f"{session_memory.to_yaml()}\n"
113
+ "==================================================\n\n"
114
+ )
115
+
116
+ system_prompt = self._system_prompt_for(planning_mode)
117
+ user_content = ""
118
+ if memory_context:
119
+ user_content += memory_context
120
+ if research_context:
121
+ user_content += research_context
122
+ user_content += prompt
123
+
124
+ messages = [
125
+ Message(role="system", content=system_prompt),
126
+ Message(role="user", content=user_content),
127
+ ]
128
+
129
+ # 1. Generate initial draft plan (v1)
130
+ response = self.llm.generate(messages)
131
+ final_itinerary = response.text
132
+
133
+ # 2. Invoke reflection loop if a reflection agent is set
134
+ reflection_critique = None
135
+ mcp_validation: Optional[Dict[str, Any]] = None
136
+ revision_triggered = False
137
+ if self.reflection_agent:
138
+ if hasattr(self.reflection_agent, "reflect"):
139
+ reflection_output = self.reflection_agent.reflect(prompt, final_itinerary, session_memory)
140
+ else:
141
+ reflection_output = self.reflection_agent.run(
142
+ f"Scenario: {prompt}\n\nDraft Itinerary:\n{final_itinerary}"
143
+ )
144
+
145
+ critique = reflection_output.content
146
+
147
+ # v2.1 must validate every replanning draft. Reflection approval
148
+ # does not prove that hard constraints or savings arithmetic hold.
149
+ is_replanning = False
150
+ if session_memory and session_memory.state.current_day > 1:
151
+ is_replanning = True
152
+ elif "currently on day" in prompt.lower() or "replanning" in prompt.lower() or "disruption" in prompt.lower():
153
+ is_replanning = True
154
+
155
+ initial_validation = None
156
+ if is_replanning and self.validation_client and validation_scenario_id:
157
+ initial_validation = self._validate_replanning_proposal(
158
+ system_prompt=system_prompt,
159
+ scenario_prompt=prompt,
160
+ draft_itinerary=final_itinerary,
161
+ critique=critique,
162
+ scenario_id=validation_scenario_id,
163
+ )
164
+ mcp_validation = {"initial": initial_validation, "final": None}
165
+
166
+ reflection_requires_revision = "ITINERARY APPROVED" not in critique
167
+ mcp_requires_revision = self._validation_requires_revision(initial_validation)
168
+ revision_triggered = reflection_requires_revision or mcp_requires_revision
169
+
170
+ if reflection_requires_revision:
171
+ reflection_critique = critique
172
+
173
+ if revision_triggered:
174
+ # Increment plan version in memory
175
+ if session_memory:
176
+ session_memory.state.current_plan_version += 1
177
+
178
+ # Regenerate memory context with updated plan version
179
+ memory_context_v2 = ""
180
+ if session_memory:
181
+ memory_context_v2 = (
182
+ "### CURRENT TRAVELER SESSION STATE (YAML format):\n"
183
+ f"{session_memory.to_yaml()}\n"
184
+ "==================================================\n\n"
185
+ )
186
+
187
+ if is_replanning:
188
+ revision_rules = (
189
+ "1. PRESERVE every part of the itinerary that was NOT criticized. Do NOT rewrite or shift unaffected days.\n"
190
+ "2. Do NOT modify, delete, or rearrange locked bookings (e.g., booked flights, non-refundable hotel stays).\n"
191
+ "3. Make ONLY the minimum necessary adjustments needed to resolve the critique points."
192
+ )
193
+ else:
194
+ revision_rules = (
195
+ "1. You are free to re-sequence, compress, or optimize the entire itinerary globally to satisfy the constraints (e.g., total duration, must-visit destinations) and address the critiques.\n"
196
+ "2. Do NOT exceed the total duration limit specified in the original request.\n"
197
+ "3. Make adjustments to address the critiques while keeping the travel style and preferences consistent."
198
+ )
199
+
200
+ # Formulate revision prompt
201
+ validation_context = ""
202
+ if initial_validation:
203
+ validation_context = (
204
+ "\n\n### DETERMINISTIC MCP VALIDATION REPORT:\n"
205
+ f"{json.dumps(initial_validation, indent=2)}\n"
206
+ "Treat every violation or unmet savings target in this report as a hard "
207
+ "requirement for the final revision. Do not claim a saving unless it is "
208
+ "represented by a concrete itinerary change.\n"
209
+ )
210
+ revision_prompt = (
211
+ f"### ORIGINAL SCENARIO REQUEST:\n{prompt}\n\n"
212
+ f"### INITIAL DRAFT PLAN:\n{final_itinerary}\n\n"
213
+ f"### REFLECTION CRITIQUE:\n{critique}\n\n"
214
+ "You are instructed to revise the initial draft plan to resolve the critiques listed above.\n"
215
+ "You MUST strictly follow these rules during the revision:\n"
216
+ f"{revision_rules}\n"
217
+ "4. If a critique point directly conflicts with an existing hard constraint in the original request, prioritize and preserve the hard constraint.\n"
218
+ "5. You MUST generate a complete, full day-by-day traveler-facing itinerary. Do NOT output a list of questions, deferrals, or requests for information in place of the itinerary."
219
+ f"{validation_context}"
220
+ )
221
+
222
+ user_content_v2 = ""
223
+ if memory_context_v2:
224
+ user_content_v2 += memory_context_v2
225
+ if research_context:
226
+ user_content_v2 += research_context
227
+ user_content_v2 += revision_prompt
228
+
229
+ messages_v2 = [
230
+ Message(role="system", content=system_prompt),
231
+ Message(role="user", content=user_content_v2),
232
+ ]
233
+ # Generate revised plan (v2)
234
+ response = self.llm.generate(messages_v2)
235
+ if response.text.strip():
236
+ final_itinerary = response.text
237
+
238
+ if is_replanning and self.validation_client and validation_scenario_id:
239
+ mcp_validation["final"] = self._validate_replanning_proposal(
240
+ system_prompt=system_prompt,
241
+ scenario_prompt=prompt,
242
+ draft_itinerary=final_itinerary,
243
+ critique="Validate the final revised itinerary against the original constraints.",
244
+ scenario_id=validation_scenario_id,
245
+ )
246
+
247
+ # 3. Assemble metadata
248
+ metadata = {
249
+ "agent": self.__class__.__name__,
250
+ "llm": type(self.llm).__name__,
251
+ "reflection_approved": not bool(reflection_critique),
252
+ "revision_triggered": revision_triggered,
253
+ }
254
+ if research_metadata:
255
+ metadata["research_steps"] = research_metadata
256
+ if revision_triggered:
257
+ metadata["reflection_critique"] = reflection_critique
258
+ metadata["final_plan_version"] = session_memory.state.current_plan_version if session_memory else 2
259
+ if mcp_validation:
260
+ metadata["mcp_validation"] = mcp_validation
261
+
262
+ return AgentOutput(
263
+ content=final_itinerary,
264
+ metadata=metadata,
265
+ )
266
+
267
+ def _validate_replanning_proposal(
268
+ self,
269
+ system_prompt: str,
270
+ scenario_prompt: str,
271
+ draft_itinerary: str,
272
+ critique: str,
273
+ scenario_id: str,
274
+ ) -> Dict[str, Any]:
275
+ """Gets explicit revision claims and validates them through MCP.
276
+
277
+ The model does not use this call to write the final traveler-facing
278
+ answer. It only declares the concrete changes it intends to make, which
279
+ lets MCP verify locked bookings and arithmetic before the final rewrite.
280
+ A malformed summary or unavailable server is reported to the final
281
+ revision prompt rather than silently treated as a passing check.
282
+ """
283
+ # Retrieve exact locked constraints to ground the model on valid booking IDs
284
+ locked_info = {}
285
+ if self.validation_client:
286
+ try:
287
+ locked_info = self._call_validation_tool(
288
+ "get_locked_constraints", {"scenario_id": scenario_id}
289
+ )
290
+ except Exception:
291
+ pass
292
+
293
+ immutable_ids = locked_info.get("immutable_booking_ids", ["kyoto-hostel", "narita-return-flight"])
294
+
295
+ summary_prompt = (
296
+ "Return only one JSON object describing the proposed replanning revision. "
297
+ "Do not write an itinerary and do not use Markdown outside the JSON.\n\n"
298
+ f"The scenario_id must be exactly: {scenario_id}\n"
299
+ f"The exact booking_id values to preserve in booking_actions MUST be: {json.dumps(immutable_ids)}. "
300
+ "Do NOT add suffixes or alter these booking_id strings.\n"
301
+ "List every itinerary day you intend to change. For every locked booking, "
302
+ "declare preserve, move, cancel, or modify. Itemize each concrete saving "
303
+ "in INR; do not estimate a total without its components.\n\n"
304
+ f"Required JSON shape:\n{REVISION_SUMMARY_JSON_TEMPLATE}\n\n"
305
+ f"### ORIGINAL SCENARIO REQUEST:\n{scenario_prompt}\n\n"
306
+ f"### INITIAL DRAFT PLAN:\n{draft_itinerary}\n\n"
307
+ f"### REFLECTION CRITIQUE:\n{critique}\n"
308
+ )
309
+ summary_response = self.llm.generate(
310
+ [
311
+ Message(role="system", content=system_prompt),
312
+ Message(role="user", content=summary_prompt),
313
+ ]
314
+ )
315
+
316
+ try:
317
+ summary = parse_revision_summary(summary_response.text)
318
+ if summary.scenario_id != scenario_id:
319
+ raise ValueError(
320
+ "Revision summary scenario_id does not match the requested scenario."
321
+ )
322
+ tool_arguments = summary.to_tool_arguments()
323
+ return {
324
+ "status": "completed",
325
+ "revision_summary": tool_arguments,
326
+ "locked_constraints": locked_info or self._call_validation_tool(
327
+ "get_locked_constraints", {"scenario_id": scenario_id}
328
+ ),
329
+ "revision_check": self._call_validation_tool(
330
+ "validate_revision",
331
+ {
332
+ "scenario_id": scenario_id,
333
+ "booking_actions": tool_arguments["booking_actions"],
334
+ "changed_days": tool_arguments["changed_days"],
335
+ },
336
+ ),
337
+ "savings_check": self._call_validation_tool(
338
+ "calculate_savings",
339
+ {
340
+ "scenario_id": scenario_id,
341
+ "savings_items": tool_arguments["savings_items"],
342
+ },
343
+ ),
344
+ }
345
+ except (RuntimeError, ValueError) as error:
346
+ return {
347
+ "status": "unavailable",
348
+ "error": str(error),
349
+ "raw_revision_summary": summary_response.text,
350
+ }
351
+
352
+ def _call_validation_tool(self, name: str, arguments: Dict[str, Any]) -> Dict[str, Any]:
353
+ """Runs an async MCP call from this currently synchronous agent API."""
354
+ return asyncio.run(self.validation_client.call_tool(name, arguments))
355
+
356
+ @staticmethod
357
+ def _validation_requires_revision(validation: Optional[Dict[str, Any]]) -> bool:
358
+ """Returns whether an MCP result requires a constrained revision."""
359
+ if validation is None:
360
+ return False
361
+ if validation.get("status") != "completed":
362
+ return True
363
+ return (
364
+ not validation.get("revision_check", {}).get("valid", False)
365
+ or not validation.get("savings_check", {}).get("target_met", False)
366
+ )
367
+
368
+ @staticmethod
369
+ def _system_prompt_for(planning_mode: str) -> str:
370
+ """Returns the planner policy for normal or controlled benchmark runs."""
371
+ if planning_mode == "standard":
372
+ return TRAVEL_PLANNING_SYSTEM_PROMPT
373
+ if planning_mode == "closed_world_evaluation":
374
+ return TRAVEL_PLANNING_SYSTEM_PROMPT + "\n" + CLOSED_WORLD_EVALUATION_INSTRUCTION
375
+ raise ValueError(
376
+ "planning_mode must be 'standard' or 'closed_world_evaluation'."
377
+ )
@@ -0,0 +1,30 @@
1
+ TRAVEL_PLANNING_SYSTEM_PROMPT = (
2
+ "You are a professional travel agent assistant.\n"
3
+ "Your task is to plan a detailed, well-structured travel itinerary "
4
+ "based on the user's preferences, budget limits, duration, backpacking style, "
5
+ "and other specified constraints.\n"
6
+ "Be precise, geographically efficient, and mention exact travel logistics, closed days, and pricing "
7
+ "where appropriate.\n\n"
8
+ "CRITICAL RULE ON UNCERTAINTY:\n"
9
+ "If the user's request lacks critical details needed to build a realistic itinerary (such as travel dates, "
10
+ "trip duration, specific destinations, or seasonal timing dependencies like cherry blossom bloom forecasts), "
11
+ "you must NOT guess, assume, or hallucinate a sample or final itinerary. Instead, you must:\n"
12
+ "1. Defer generating the final day-by-day itinerary.\n"
13
+ "2. Ask the user clarifying questions to obtain the missing details (e.g., travel style, departure airport).\n"
14
+ "3. Output structured tool requests in a YAML block specifying what external search queries are required to gather "
15
+ "the necessary data first (forecasts, festival dates, pricing). Use this exact format:\n\n"
16
+ "Tool Requests:\n"
17
+ " - Search:\n"
18
+ " query: \"<search query>\"\n"
19
+ )
20
+
21
+
22
+ CLOSED_WORLD_EVALUATION_INSTRUCTION = (
23
+ "EVALUATION SCENARIO MODE:\n"
24
+ "The scenario supplied in the user message is a complete, closed-world test "
25
+ "case. Produce a concrete provisional itinerary using its stated days, "
26
+ "constraints, and disruptions. Do not ask clarifying questions or defer the "
27
+ "itinerary merely because calendar dates, live availability, or exact prices "
28
+ "are absent. Do not invent external facts: label any operational detail that "
29
+ "would require confirmation as conditional, while still completing the plan.\n"
30
+ )
@@ -0,0 +1,110 @@
1
+ """Travel-specific session memory implementation."""
2
+
3
+ from dataclasses import dataclass, field, asdict
4
+ from typing import Any, Dict, List, Optional
5
+ import yaml
6
+ import json
7
+ from framework.memory import BaseSessionMemory
8
+
9
+
10
+ @dataclass
11
+ class Booking:
12
+ """Represents a structured locked booking."""
13
+ booking_id: str
14
+ booking_type: str # e.g., "flight", "hotel", "activity", "train"
15
+ location: str
16
+ start_day: int
17
+ end_day: int
18
+ refundable: bool = True
19
+ notes: str = ""
20
+
21
+
22
+ @dataclass
23
+ class RemoteWorkSchedule:
24
+ """Represents a remote work commitment and timezone constraints."""
25
+ timezone: str = "UTC"
26
+ availability_start: str = "09:00"
27
+ availability_end: str = "17:00"
28
+ meeting_start: str = ""
29
+ meeting_end: str = ""
30
+ flexible_hours: int = 0
31
+
32
+
33
+ @dataclass
34
+ class UserPreferences:
35
+ """Represents purely static user profiling and travel preferences."""
36
+ interests: List[str] = field(default_factory=list)
37
+ travel_style: str = "balanced"
38
+ accommodation_preference: str = "hotel"
39
+ walking_tolerance: str = "moderate"
40
+
41
+
42
+ @dataclass
43
+ class TripConstraints:
44
+ """Represents hard trip limits, schedules, and locked bookings."""
45
+ total_budget: float = 0.0
46
+ duration_days: int = 1
47
+ destinations: List[str] = field(default_factory=list)
48
+ remote_work_schedule: Optional[RemoteWorkSchedule] = None
49
+ locked_bookings: List[Booking] = field(default_factory=list)
50
+ hard_constraints: List[str] = field(default_factory=list)
51
+
52
+
53
+ @dataclass
54
+ class CurrentTripState:
55
+ """Tracks active planning state and execution progress."""
56
+ current_day: int = 1
57
+ remaining_budget: float = 0.0
58
+ completed_destinations: List[str] = field(default_factory=list)
59
+ current_plan_version: int = 1
60
+
61
+
62
+ class TravelSessionMemory(BaseSessionMemory):
63
+ """Holds and serializes working session memory for travel planning and reflection."""
64
+
65
+ def __init__(
66
+ self,
67
+ session_id: str,
68
+ version: int = 1,
69
+ preferences: Optional[UserPreferences] = None,
70
+ constraints: Optional[TripConstraints] = None,
71
+ state: Optional[CurrentTripState] = None,
72
+ ):
73
+ """Initializes the TravelSessionMemory.
74
+
75
+ Args:
76
+ session_id: A unique string identifier for the planning session.
77
+ version: Schema/memory state version.
78
+ preferences: Optional initial UserPreferences.
79
+ constraints: Optional initial TripConstraints.
80
+ state: Optional initial CurrentTripState.
81
+ """
82
+ self.session_id = session_id
83
+ self.version = version
84
+ self.preferences = preferences or UserPreferences()
85
+ self.constraints = constraints or TripConstraints()
86
+ self.state = state or CurrentTripState()
87
+
88
+ def to_yaml(self) -> str:
89
+ """Serializes session state to structured YAML for LLM context injection.
90
+
91
+ Excludes empty lists or empty dicts to conserve context tokens.
92
+ """
93
+ data = {
94
+ "session_id": self.session_id,
95
+ "version": self.version,
96
+ "preferences": {k: v for k, v in asdict(self.preferences).items() if v},
97
+ "constraints": {k: v for k, v in asdict(self.constraints).items() if v},
98
+ "state": {k: v for k, v in asdict(self.state).items() if v},
99
+ }
100
+ return yaml.dump(data, sort_keys=False, default_flow_style=False)
101
+
102
+ def to_json(self) -> str:
103
+ """Returns JSON representation of memory state."""
104
+ return json.dumps({
105
+ "session_id": self.session_id,
106
+ "version": self.version,
107
+ "preferences": asdict(self.preferences),
108
+ "constraints": asdict(self.constraints),
109
+ "state": asdict(self.state),
110
+ }, indent=2)
cli/__init__.py ADDED
@@ -0,0 +1,6 @@
1
+ """evalrun CLI package exports."""
2
+
3
+ from cli.main import main, create_parser
4
+ from cli.resolver import resolve_agent
5
+
6
+ __all__ = ["main", "create_parser", "resolve_agent"]
cli/demo.py ADDED
@@ -0,0 +1,47 @@
1
+ """Offline, zero-credential EvalRun product demonstration."""
2
+
3
+ import json
4
+ from datetime import datetime, timezone
5
+ from pathlib import Path
6
+
7
+ from cli.formatter import format_terminal_summary
8
+ from cli.html_reporter import generate_html_report
9
+ from framework.models import DimensionScore, EvaluationResult
10
+
11
+
12
+ def run_demo(output_dir: str = "results/demo") -> int:
13
+ """Create a deterministic sample report without contacting a model endpoint."""
14
+ out = Path(output_dir)
15
+ out.mkdir(parents=True, exist_ok=True)
16
+ result = EvaluationResult(
17
+ benchmark_id="evalrun-demo-scenario",
18
+ benchmark_name="EvalRun Offline Demo",
19
+ overall_score=88.0,
20
+ dimension_scores=[
21
+ DimensionScore("Constraint Satisfaction", 90.0, "All demo constraints were satisfied."),
22
+ DimensionScore("Planning Quality", 86.0, "The demo output is coherent and complete."),
23
+ ],
24
+ passed=True,
25
+ agent_metadata={"demo": True, "audit_gate_decision": "PASS"},
26
+ )
27
+ manifest = {
28
+ "run_id": "evalrun-offline-demo",
29
+ "timestamp_utc": datetime.now(timezone.utc).isoformat(),
30
+ "target_agent_spec": "built-in offline demo",
31
+ "target_model": {"model_name": "offline-demo", "base_url": "local", "api_key": "EMPTY"},
32
+ "judge_model": {"model_name": "offline-demo", "base_url": "local", "api_key": "EMPTY"},
33
+ "output_dir": str(out),
34
+ "total_scenarios": 1,
35
+ "overall_passed": True,
36
+ "scenarios": [{"scenario_id": result.benchmark_id, "scenario_name": result.benchmark_name,
37
+ "overall_score": result.overall_score, "passed": result.passed,
38
+ "audit_gate_decision": "PASS"}],
39
+ }
40
+ with open(out / "manifest.json", "w", encoding="utf-8") as handle:
41
+ json.dump(manifest, handle, indent=2)
42
+ with open(out / "evalrun-demo-output.txt", "w", encoding="utf-8") as handle:
43
+ handle.write("This is a deterministic offline preview. No model endpoint was contacted.\n")
44
+ generate_html_report([result], manifest, str(out))
45
+ print(format_terminal_summary([result], manifest))
46
+ print(f"\nOffline demo artifacts: {out.resolve()}")
47
+ return 0