evalrun 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. agents/__init__.py +6 -0
  2. agents/auditor/__init__.py +13 -0
  3. agents/auditor/budget_auditor.py +91 -0
  4. agents/auditor/parser.py +139 -0
  5. agents/auditor/prompts.py +64 -0
  6. agents/auditor/schema.py +67 -0
  7. agents/base.py +20 -0
  8. agents/reflection/__init__.py +3 -0
  9. agents/reflection/agent.py +77 -0
  10. agents/reflection/prompts.py +18 -0
  11. agents/research/__init__.py +4 -0
  12. agents/research/agent.py +45 -0
  13. agents/research/planner.py +52 -0
  14. agents/research/prompts.py +14 -0
  15. agents/support/__init__.py +5 -0
  16. agents/support/triage_agent.py +45 -0
  17. agents/travel/__init__.py +11 -0
  18. agents/travel/agent.py +377 -0
  19. agents/travel/prompts.py +30 -0
  20. agents/travel/session.py +110 -0
  21. cli/__init__.py +6 -0
  22. cli/demo.py +47 -0
  23. cli/formatter.py +93 -0
  24. cli/html_reporter.py +647 -0
  25. cli/main.py +423 -0
  26. cli/progress.py +38 -0
  27. cli/resolver.py +99 -0
  28. evalrun-0.4.0.dist-info/METADATA +268 -0
  29. evalrun-0.4.0.dist-info/RECORD +100 -0
  30. evalrun-0.4.0.dist-info/WHEEL +5 -0
  31. evalrun-0.4.0.dist-info/entry_points.txt +2 -0
  32. evalrun-0.4.0.dist-info/licenses/LICENSE +0 -0
  33. evalrun-0.4.0.dist-info/top_level.txt +4 -0
  34. framework/__init__.py +70 -0
  35. framework/core/__init__.py +17 -0
  36. framework/core/adapters.py +118 -0
  37. framework/core/contracts.py +88 -0
  38. framework/core/suite.py +44 -0
  39. framework/evaluation/__init__.py +22 -0
  40. framework/evaluation/base.py +29 -0
  41. framework/evaluation/dimensions.py +7 -0
  42. framework/evaluation/engine.py +110 -0
  43. framework/evaluation/evaluators/__init__.py +7 -0
  44. framework/evaluation/evaluators/adaptability.py +27 -0
  45. framework/evaluation/evaluators/base_llm.py +104 -0
  46. framework/evaluation/evaluators/constraint.py +27 -0
  47. framework/evaluation/evaluators/information_accuracy.py +41 -0
  48. framework/evaluation/evaluators/personalization.py +27 -0
  49. framework/evaluation/evaluators/planning.py +27 -0
  50. framework/evaluation/evaluators/support.py +81 -0
  51. framework/evaluation/prompts/__init__.py +11 -0
  52. framework/evaluation/prompts/adaptability.py +57 -0
  53. framework/evaluation/prompts/base.py +52 -0
  54. framework/evaluation/prompts/constraint.py +41 -0
  55. framework/evaluation/prompts/information_accuracy.py +79 -0
  56. framework/evaluation/prompts/personalization.py +57 -0
  57. framework/evaluation/prompts/planning.py +61 -0
  58. framework/evaluation/runner.py +363 -0
  59. framework/evaluation/testing.py +25 -0
  60. framework/exceptions.py +49 -0
  61. framework/llms/__init__.py +8 -0
  62. framework/llms/base.py +37 -0
  63. framework/llms/factory.py +38 -0
  64. framework/llms/gemini.py +85 -0
  65. framework/llms/mock.py +25 -0
  66. framework/llms/openai.py +94 -0
  67. framework/llms/openai_compatible.py +139 -0
  68. framework/mcp/__init__.py +20 -0
  69. framework/mcp/client.py +62 -0
  70. framework/mcp/constraints.py +125 -0
  71. framework/mcp/revision_summary.py +122 -0
  72. framework/mcp/server.py +49 -0
  73. framework/memory/__init__.py +3 -0
  74. framework/memory/base.py +17 -0
  75. framework/models.py +83 -0
  76. framework/parser.py +45 -0
  77. framework/parsers/__init__.py +12 -0
  78. framework/parsers/frontmatter.py +28 -0
  79. framework/parsers/mapper.py +66 -0
  80. framework/parsers/markdown.py +122 -0
  81. framework/parsers/transformers.py +112 -0
  82. framework/profiles/__init__.py +28 -0
  83. framework/profiles/registry.py +104 -0
  84. framework/profiles/support.py +27 -0
  85. framework/profiles/travel.py +78 -0
  86. framework/regression/__init__.py +17 -0
  87. framework/regression/comparator.py +273 -0
  88. framework/regression/loader.py +145 -0
  89. framework/sdk.py +151 -0
  90. framework/utils.py +41 -0
  91. framework/verification/__init__.py +13 -0
  92. framework/verification/base.py +32 -0
  93. framework/verification/extractor.py +105 -0
  94. framework/verification/local.py +122 -0
  95. framework/verification/models.py +112 -0
  96. framework/verification/pipeline.py +36 -0
  97. framework/verification/prompts.py +24 -0
  98. framework/verification/utils.py +54 -0
  99. ui/__init__.py +1 -0
  100. ui/server.py +255 -0
agents/__init__.py ADDED
@@ -0,0 +1,6 @@
1
+ """Agents package representing subjects under evaluation."""
2
+
3
+ from agents.base import BaseAgent
4
+ from agents.research.agent import ResearchAgent
5
+ from agents.research.planner import ResearchPlanner
6
+ from agents.reflection.agent import ReflectionAgent
@@ -0,0 +1,13 @@
1
+ """Independent Budget Auditor package exports."""
2
+
3
+ from agents.auditor.schema import AuditReport, BudgetViolation, FailureCode
4
+ from agents.auditor.parser import AuditParser
5
+ from agents.auditor.budget_auditor import IndependentBudgetAuditor
6
+
7
+ __all__ = [
8
+ "AuditReport",
9
+ "AuditParser",
10
+ "BudgetViolation",
11
+ "FailureCode",
12
+ "IndependentBudgetAuditor",
13
+ ]
@@ -0,0 +1,91 @@
1
+ """Independent Budget Auditor implementation for financial verification."""
2
+
3
+ from typing import Optional, Any
4
+
5
+ from agents.base import BaseAgent
6
+ from agents.auditor.schema import AuditReport, BudgetViolation, FailureCode
7
+ from agents.auditor.prompts import AUDITOR_SYSTEM_PROMPT, format_auditor_user_prompt
8
+ from agents.auditor.parser import AuditParser
9
+ from framework.llms.base import BaseLLM, Message
10
+
11
+
12
+ class IndependentBudgetAuditor(BaseAgent):
13
+ """An independent budget auditing agent with zero shared memory of reflection state."""
14
+
15
+ def __init__(self, llm: BaseLLM, max_retries: int = 2):
16
+ self.llm = llm
17
+ self.max_retries = max_retries
18
+
19
+ def audit(
20
+ self,
21
+ scenario_prompt: str,
22
+ itinerary_content: str,
23
+ total_budget_inr: Optional[float] = None,
24
+ daily_budget_jpy: Optional[float] = None,
25
+ ) -> AuditReport:
26
+ """Independently audits an itinerary against scenario budget constraints.
27
+
28
+ Args:
29
+ scenario_prompt: Original scenario specification prompt.
30
+ itinerary_content: Finalized itinerary content to audit.
31
+ total_budget_inr: Optional overall total budget limit in INR.
32
+ daily_budget_jpy: Optional daily spend allowance limit in JPY.
33
+
34
+ Returns:
35
+ An AuditReport instance detailing PASS/BLOCK status and violations.
36
+ """
37
+ user_content = format_auditor_user_prompt(
38
+ scenario_prompt=scenario_prompt,
39
+ itinerary_content=itinerary_content,
40
+ total_budget_inr=total_budget_inr,
41
+ daily_budget_jpy=daily_budget_jpy,
42
+ )
43
+
44
+ messages = [
45
+ Message(role="system", content=AUDITOR_SYSTEM_PROMPT),
46
+ Message(role="user", content=user_content),
47
+ ]
48
+
49
+ last_error = ""
50
+ last_raw_response = ""
51
+
52
+ for attempt in range(self.max_retries + 1):
53
+ response = self.llm.generate(messages)
54
+ last_raw_response = response.text
55
+ report, parse_err = AuditParser.parse_and_validate(response.text)
56
+ if report is not None:
57
+ report.retries_attempted = attempt
58
+ return report
59
+
60
+ last_error = parse_err or "Unknown validation error"
61
+ if attempt < self.max_retries:
62
+ messages.append(Message(role="assistant", content=response.text))
63
+ messages.append(
64
+ Message(
65
+ role="user",
66
+ content=f"Your previous response was rejected due to validation failure: '{last_error}'. Please correct your output and return ONLY a valid JSON object matching the exact schema inside ```json ... ```.",
67
+ )
68
+ )
69
+
70
+ # Fallback default report if all retries fail
71
+ return AuditReport(
72
+ status="BLOCK",
73
+ audit_score=0.0,
74
+ violations=[
75
+ BudgetViolation(
76
+ violation_type=FailureCode.MATH_HALLUCINATION,
77
+ description=f"Auditor failed output validation: {last_error}",
78
+ )
79
+ ],
80
+ reasoning_summary=f"Audit failed due to model output validation failure: {last_error}",
81
+ audit_confidence=0.0,
82
+ parse_error=last_error,
83
+ retries_attempted=self.max_retries + 1,
84
+ raw_model_response=last_raw_response,
85
+ )
86
+
87
+ def run(self, prompt: str, **kwargs) -> Any:
88
+ """Standard AgentOutput wrapper for pipeline runner compatibility."""
89
+ itinerary = kwargs.get("itinerary_content", prompt)
90
+ scenario = kwargs.get("scenario_prompt", prompt)
91
+ return self.audit(scenario_prompt=scenario, itinerary_content=itinerary)
@@ -0,0 +1,139 @@
1
+ """Structured JSON output parser and validator for AuditReport."""
2
+
3
+ import json
4
+ import re
5
+ from typing import Optional, Tuple, List, Dict, Any
6
+
7
+ from agents.auditor.schema import AuditReport, BudgetViolation, FailureCode
8
+
9
+
10
+ class AuditParser:
11
+ """Parser and validator for Independent Budget Auditor LLM responses."""
12
+
13
+ @staticmethod
14
+ def parse_and_validate(text: str) -> Tuple[Optional[AuditReport], Optional[str]]:
15
+ """Strictly parses and validates AuditReport from raw LLM output.
16
+
17
+ Args:
18
+ text: Raw response string from the model.
19
+
20
+ Returns:
21
+ Tuple of (AuditReport, None) on success, or (None, error_message) on failure.
22
+ """
23
+ try:
24
+ # 1. Extract JSON block or brace contents
25
+ match = re.search(r"```json\s*(\{.*?\})\s*```", text, re.DOTALL)
26
+ if match:
27
+ raw_json = match.group(1)
28
+ else:
29
+ match_brace = re.search(r"(\{.*\})", text, re.DOTALL)
30
+ if match_brace:
31
+ raw_json = match_brace.group(1)
32
+ else:
33
+ return None, "No JSON codeblock or object found in model output"
34
+
35
+ # 2. Parse JSON syntax
36
+ try:
37
+ data = json.loads(raw_json)
38
+ except Exception as ex:
39
+ return None, f"JSON syntax error: {str(ex)}"
40
+
41
+ if not isinstance(data, dict):
42
+ return None, "Root JSON payload must be an object"
43
+
44
+ # 3. Check mandatory keys
45
+ required_keys = [
46
+ "status",
47
+ "audit_score",
48
+ "violations",
49
+ "total_estimated_spend_inr",
50
+ "budget_limit_inr",
51
+ "variance_inr",
52
+ "audit_confidence",
53
+ "reasoning_summary",
54
+ ]
55
+ missing_keys = [k for k in required_keys if k not in data]
56
+ if missing_keys:
57
+ return None, f"Missing required top-level JSON keys: {missing_keys}"
58
+
59
+ # 4. Validate status
60
+ status = str(data["status"]).upper().strip()
61
+ if status not in ("PASS", "BLOCK"):
62
+ return None, f"Invalid status '{data['status']}'; must be 'PASS' or 'BLOCK'"
63
+
64
+ # 5. Validate numeric ranges
65
+ try:
66
+ audit_score = float(data["audit_score"])
67
+ if not (0.0 <= audit_score <= 100.0):
68
+ return None, f"audit_score {audit_score} out of bounds [0.0, 100.0]"
69
+ except (ValueError, TypeError):
70
+ return None, "audit_score must be a numeric float"
71
+
72
+ try:
73
+ audit_confidence = float(data["audit_confidence"])
74
+ if not (0.0 <= audit_confidence <= 1.0):
75
+ return None, f"audit_confidence {audit_confidence} out of bounds [0.0, 1.0]"
76
+ except (ValueError, TypeError):
77
+ return None, "audit_confidence must be a numeric float"
78
+
79
+ # 6. Validate violations list and failure codes
80
+ raw_violations = data["violations"]
81
+ if not isinstance(raw_violations, list):
82
+ return None, "'violations' must be a JSON array"
83
+
84
+ violations: List[BudgetViolation] = []
85
+ for idx, v in enumerate(raw_violations):
86
+ if not isinstance(v, dict):
87
+ return None, f"violation at index {idx} must be an object"
88
+
89
+ v_type = v.get("violation_type")
90
+ if not v_type or v_type not in FailureCode.ALL_CODES:
91
+ return None, f"violation at index {idx} has invalid type '{v_type}'; must be one of {FailureCode.ALL_CODES}"
92
+
93
+ description = v.get("description")
94
+ if not description or not str(description).strip():
95
+ return None, f"violation at index {idx} is missing a description"
96
+
97
+ try:
98
+ disc = float(v.get("estimated_discrepancy_inr", 0.0))
99
+ except (ValueError, TypeError):
100
+ disc = 0.0
101
+
102
+ affected_days = []
103
+ if "affected_days" in v and isinstance(v["affected_days"], list):
104
+ for d in v["affected_days"]:
105
+ try:
106
+ affected_days.append(int(d))
107
+ except (ValueError, TypeError):
108
+ pass
109
+
110
+ violations.append(
111
+ BudgetViolation(
112
+ violation_type=v_type,
113
+ description=str(description).strip(),
114
+ estimated_discrepancy_inr=disc,
115
+ affected_days=affected_days,
116
+ )
117
+ )
118
+
119
+ # 7. Validate Status vs Violations Consistency
120
+ if status == "PASS" and len(violations) > 0:
121
+ return None, f"Status is 'PASS' but violations array is non-empty ({len(violations)} violations specified)"
122
+ if status == "BLOCK" and len(violations) == 0:
123
+ return None, "Status is 'BLOCK' but violations array is empty (must list at least 1 violation)"
124
+
125
+ return (
126
+ AuditReport(
127
+ status=status,
128
+ audit_score=audit_score,
129
+ violations=violations,
130
+ total_estimated_spend_inr=float(data.get("total_estimated_spend_inr", 0.0)),
131
+ budget_limit_inr=float(data.get("budget_limit_inr", 0.0)),
132
+ variance_inr=float(data.get("variance_inr", 0.0)),
133
+ audit_confidence=audit_confidence,
134
+ reasoning_summary=str(data.get("reasoning_summary", "")),
135
+ ),
136
+ None,
137
+ )
138
+ except Exception as e:
139
+ return None, f"Unexpected error parsing audit JSON: {str(e)}"
@@ -0,0 +1,64 @@
1
+ """System and user prompts for the Independent Budget Auditor."""
2
+
3
+ AUDITOR_SYSTEM_PROMPT = """You are an Independent Financial Budget Auditor for travel itineraries.
4
+ Your sole job is to independently inspect a finalized travel itinerary against the given scenario requirements and verify financial compliance.
5
+
6
+ CRITICAL RULES:
7
+ 1. You are READ-ONLY. You MUST NOT rewrite, edit, or summarize the itinerary.
8
+ 2. You MUST NOT negotiate constraints or assume unstated discounts.
9
+ 3. Be strict, precise, and objective.
10
+
11
+ You must check for four specific failure types:
12
+ 1. MATH_HALLUCINATION: The itinerary claims to save money (e.g. "taking highway bus saves ₹20,000") without providing verifiable pricing arithmetic or explicit itemized breakdowns.
13
+ 2. DAILY_OVERRUN: The sum of food, local transit, and activities on any given day exceeds the specified daily budget allowance limit.
14
+ 3. HIDDEN_OVERHEAD: Essential travel expenses (e.g. airport return transfer, luggage lockers, mandatory temple/attraction entry fees, local IC card top-ups) are omitted to artificially appear under budget.
15
+ 4. ANCHOR_MUTATION: Pre-paid, locked, or non-refundable bookings (e.g. specific hotel stays or return flights) are altered, canceled, or replaced without authorization.
16
+
17
+ OUTPUT FORMAT:
18
+ You MUST respond ONLY with a valid JSON object wrapped in ```json ... ``` with the following keys:
19
+ {
20
+ "status": "PASS" | "BLOCK",
21
+ "audit_score": <float between 0.0 and 100.0>,
22
+ "violations": [
23
+ {
24
+ "violation_type": "MATH_HALLUCINATION" | "DAILY_OVERRUN" | "HIDDEN_OVERHEAD" | "ANCHOR_MUTATION",
25
+ "description": "<detailed explanation of the violation>",
26
+ "estimated_discrepancy_inr": <estimated discrepancy in INR float>,
27
+ "affected_days": [<list of integer day numbers>]
28
+ }
29
+ ],
30
+ "total_estimated_spend_inr": <float>,
31
+ "budget_limit_inr": <float>,
32
+ "variance_inr": <float, positive means under budget, negative means overspend>,
33
+ "audit_confidence": <float between 0.0 and 1.0>,
34
+ "reasoning_summary": "<brief summary of findings>"
35
+ }
36
+ """
37
+
38
+
39
+ def format_auditor_user_prompt(
40
+ scenario_prompt: str,
41
+ itinerary_content: str,
42
+ total_budget_inr: float = None,
43
+ daily_budget_jpy: float = None,
44
+ ) -> str:
45
+ """Formats the user input text for the auditor.
46
+
47
+ Args:
48
+ scenario_prompt: Scenario specification prompt.
49
+ itinerary_content: Finalized itinerary to audit.
50
+ total_budget_inr: Optional overall total budget in INR.
51
+ daily_budget_jpy: Optional daily spend allowance limit in JPY.
52
+
53
+ Returns:
54
+ Formatted user prompt string.
55
+ """
56
+ user_content = f"### SCENARIO PROMPT\n{scenario_prompt}\n\n"
57
+ if total_budget_inr is not None:
58
+ user_content += f"TOTAL BUDGET LIMIT (INR): ₹{total_budget_inr:,.2f}\n"
59
+ if daily_budget_jpy is not None:
60
+ user_content += f"DAILY ALLOWANCE LIMIT (JPY): ¥{daily_budget_jpy:,.2f}\n"
61
+
62
+ user_content += f"\n### FINAL ITINERARY TO AUDIT\n{itinerary_content}\n\n"
63
+ user_content += "Perform your independent budget audit and return JSON only."
64
+ return user_content
@@ -0,0 +1,67 @@
1
+ """Data contracts for the Independent Budget Auditor."""
2
+
3
+ from dataclasses import dataclass, field
4
+ from typing import List, Optional, Dict, Any
5
+
6
+
7
+ class FailureCode:
8
+ MATH_HALLUCINATION = "MATH_HALLUCINATION"
9
+ DAILY_OVERRUN = "DAILY_OVERRUN"
10
+ HIDDEN_OVERHEAD = "HIDDEN_OVERHEAD"
11
+ ANCHOR_MUTATION = "ANCHOR_MUTATION"
12
+
13
+ ALL_CODES = [MATH_HALLUCINATION, DAILY_OVERRUN, HIDDEN_OVERHEAD, ANCHOR_MUTATION]
14
+
15
+
16
+ @dataclass
17
+ class BudgetViolation:
18
+ violation_type: str # One of FailureCode.ALL_CODES
19
+ description: str
20
+ estimated_discrepancy_inr: float = 0.0
21
+ affected_days: List[int] = field(default_factory=list)
22
+
23
+ def to_dict(self) -> Dict[str, Any]:
24
+ return {
25
+ "violation_type": self.violation_type,
26
+ "description": self.description,
27
+ "estimated_discrepancy_inr": self.estimated_discrepancy_inr,
28
+ "affected_days": self.affected_days,
29
+ }
30
+
31
+
32
+ @dataclass
33
+ class AuditReport:
34
+ status: str # "PASS" or "BLOCK"
35
+ audit_score: float # 0.0 to 100.0
36
+ violations: List[BudgetViolation] = field(default_factory=list)
37
+ total_estimated_spend_inr: float = 0.0
38
+ budget_limit_inr: float = 0.0
39
+ variance_inr: float = 0.0
40
+ audit_confidence: float = 1.0
41
+ reasoning_summary: str = ""
42
+ parse_error: Optional[str] = None
43
+ retries_attempted: int = 0
44
+ raw_model_response: Optional[str] = None
45
+
46
+ @property
47
+ def passed(self) -> bool:
48
+ return self.status.upper() == "PASS"
49
+
50
+ def to_dict(self) -> Dict[str, Any]:
51
+ res = {
52
+ "status": self.status,
53
+ "passed": self.passed,
54
+ "audit_score": self.audit_score,
55
+ "violations": [v.to_dict() for v in self.violations],
56
+ "total_estimated_spend_inr": self.total_estimated_spend_inr,
57
+ "budget_limit_inr": self.budget_limit_inr,
58
+ "variance_inr": self.variance_inr,
59
+ "audit_confidence": self.audit_confidence,
60
+ "reasoning_summary": self.reasoning_summary,
61
+ "retries_attempted": self.retries_attempted,
62
+ }
63
+ if self.parse_error:
64
+ res["parse_error"] = self.parse_error
65
+ if self.raw_model_response:
66
+ res["raw_model_response"] = self.raw_model_response
67
+ return res
agents/base.py ADDED
@@ -0,0 +1,20 @@
1
+ """Abstract base agent class definition."""
2
+
3
+ from abc import ABC, abstractmethod
4
+ from framework.models import AgentOutput
5
+
6
+
7
+ class BaseAgent(ABC):
8
+ """Abstract base class defining the standard interface for all agents."""
9
+
10
+ @abstractmethod
11
+ def run(self, prompt: str) -> AgentOutput:
12
+ """Executes the agent's primary decision/generation task.
13
+
14
+ Args:
15
+ prompt: The user request/scenario prompt.
16
+
17
+ Returns:
18
+ An AgentOutput dataclass containing generated content and metadata.
19
+ """
20
+ pass
@@ -0,0 +1,3 @@
1
+ """Reflection agent package."""
2
+
3
+ from agents.reflection.agent import ReflectionAgent
@@ -0,0 +1,77 @@
1
+ """Reflection agent implementation for auditing and critiquing planned itineraries."""
2
+
3
+ from typing import Optional
4
+ from langfuse import observe
5
+ from framework.llms import BaseLLM, Message
6
+ from framework.models import AgentOutput
7
+ from framework.memory import BaseSessionMemory
8
+ from agents.base import BaseAgent
9
+ from .prompts import REFLECTION_SYSTEM_PROMPT
10
+
11
+
12
+ class ReflectionAgent(BaseAgent):
13
+ """An agent that critiques travel itineraries against traveler preferences and constraints."""
14
+
15
+ def __init__(self, llm: BaseLLM):
16
+ """Initializes the ReflectionAgent.
17
+
18
+ Args:
19
+ llm: The LLM client wrapper to use for auditing the itinerary.
20
+ """
21
+ self.llm = llm
22
+
23
+ @observe(name="reflection-agent")
24
+ def run(self, prompt: str) -> AgentOutput:
25
+ """Fallback wrapper to satisfy BaseAgent abstract interface.
26
+
27
+ Args:
28
+ prompt: A combined text containing request details and itinerary to critique.
29
+
30
+ Returns:
31
+ An AgentOutput containing the critique text.
32
+ """
33
+ messages = [
34
+ Message(role="system", content=REFLECTION_SYSTEM_PROMPT),
35
+ Message(role="user", content=prompt),
36
+ ]
37
+ response = self.llm.generate(messages)
38
+
39
+ return AgentOutput(
40
+ content=response.text,
41
+ metadata={
42
+ "agent": self.__class__.__name__,
43
+ "llm": type(self.llm).__name__,
44
+ },
45
+ )
46
+
47
+ def reflect(
48
+ self,
49
+ prompt: str,
50
+ itinerary: str,
51
+ session_memory: Optional[BaseSessionMemory] = None,
52
+ ) -> AgentOutput:
53
+ """Audits a travel plan and returns structured critiques.
54
+
55
+ Args:
56
+ prompt: The original user travel request/scenario prompt.
57
+ itinerary: The draft itinerary to evaluate.
58
+ session_memory: Optional active session memory context.
59
+
60
+ Returns:
61
+ An AgentOutput containing critiques or 'ITINERARY APPROVED'.
62
+ """
63
+ memory_str = ""
64
+ if session_memory:
65
+ memory_str = (
66
+ "### CURRENT TRAVELER SESSION STATE:\n"
67
+ f"{session_memory.to_yaml()}\n"
68
+ "==================================================\n\n"
69
+ )
70
+
71
+ user_content = (
72
+ f"{memory_str}"
73
+ f"### ORIGINAL USER PROMPT:\n{prompt}\n\n"
74
+ f"### DRAFT ITINERARY TO AUDIT:\n{itinerary}\n"
75
+ )
76
+
77
+ return self.run(user_content)
@@ -0,0 +1,18 @@
1
+ """System prompt instructions for the Reflection Agent."""
2
+
3
+ REFLECTION_SYSTEM_PROMPT = (
4
+ "You are a critical travel reflection assistant.\n"
5
+ "Your job is to evaluate draft travel itineraries against the user's constraints, preferences, "
6
+ "and work schedules, identifying any issues, inefficiencies, or rule violations.\n\n"
7
+ "Analyze the plan for:\n"
8
+ "1. Constraint Violations: Exceeding budget caps, duration mismatch, missing destinations.\n"
9
+ "2. Schedule Clashes: Commits during remote work/meeting hours or fails to account for timezone shifts.\n"
10
+ "3. Routing Inefficiencies: Geographic backtracking or excessive travel times.\n"
11
+ "4. Personalization Gaps: Mismatch with traveler interests, walking tolerance, or housing styles.\n\n"
12
+ "CRITICAL RULES:\n"
13
+ "- Do NOT request revisions for minor cosmetic, stylistic, or wording preferences.\n"
14
+ "- If the itinerary satisfies all hard constraints, budget limits, schedule blocks, must-visit destinations, and contains only cosmetic improvements, you MUST reply ONLY with 'ITINERARY APPROVED'.\n"
15
+ "- Only raise critiques for true, concrete violations (e.g. over-budget, meeting conflicts, missing cities, severe backtracking/routing errors).\n\n"
16
+ "Structure your feedback clearly, listing critical issues that must be corrected. "
17
+ "If the itinerary is fully correct and satisfies all constraints, reply ONLY with 'ITINERARY APPROVED'."
18
+ )
@@ -0,0 +1,4 @@
1
+ """Research agent package."""
2
+
3
+ from agents.research.agent import ResearchAgent
4
+ from agents.research.planner import ResearchPlanner
@@ -0,0 +1,45 @@
1
+ from langfuse import observe
2
+
3
+ from framework.llms import BaseLLM, Message
4
+ from framework.models import AgentOutput
5
+ from agents.base import BaseAgent
6
+ from .prompts import RESEARCH_SYSTEM_PROMPT
7
+
8
+
9
+ class ResearchAgent(BaseAgent):
10
+ """A travel research assistant agent designed to retrieve factual answers."""
11
+
12
+ def __init__(self, llm: BaseLLM):
13
+ """Initializes the ResearchAgent.
14
+
15
+ Args:
16
+ llm: The LLM client wrapper to use for reasoning and answering questions.
17
+ """
18
+ self.llm = llm
19
+
20
+ @observe(name="research-agent")
21
+ def run(self, prompt: str) -> AgentOutput:
22
+ """Runs the research agent to answer a specific factual query.
23
+
24
+ Args:
25
+ prompt: The specific question/query (e.g. transit time, price, schedule).
26
+
27
+ Returns:
28
+ An AgentOutput containing the research answer.
29
+ """
30
+ messages = [
31
+ Message(role="system", content=RESEARCH_SYSTEM_PROMPT),
32
+ Message(role="user", content=prompt),
33
+ ]
34
+
35
+ response = self.llm.generate(messages)
36
+
37
+ metadata = {
38
+ "agent": self.__class__.__name__,
39
+ "llm": type(self.llm).__name__,
40
+ }
41
+
42
+ return AgentOutput(
43
+ content=response.text,
44
+ metadata=metadata,
45
+ )
@@ -0,0 +1,52 @@
1
+ """Research planner for generating travel fact queries to research."""
2
+
3
+ from framework.llms import BaseLLM, Message
4
+ from framework.utils import parse_json_markdown
5
+
6
+ from langfuse import observe
7
+
8
+
9
+ class ResearchPlanner:
10
+ """Decides what factual details need to be researched based on a user's travel request."""
11
+
12
+ def __init__(self, llm: BaseLLM):
13
+ """Initializes the ResearchPlanner.
14
+
15
+ Args:
16
+ llm: The LLM client wrapper to use for generating queries.
17
+ """
18
+ self.llm = llm
19
+
20
+ @observe(name="research-planner")
21
+ def plan_queries(self, prompt: str) -> list[str]:
22
+ """Generates up to 3 factual queries to guide research.
23
+
24
+ Args:
25
+ prompt: The user request/scenario prompt.
26
+
27
+ Returns:
28
+ A list of query strings.
29
+ """
30
+ system_instruction = (
31
+ "You are a helpful travel planning assistant.\n"
32
+ "Given the user's travel request, identify up to 3 critical factual questions "
33
+ "(e.g., transit times between cities, lodging prices, attraction opening hours or closures, "
34
+ "seasonal forecasts) that should be researched first to make the itinerary realistic.\n"
35
+ "Respond ONLY with a valid JSON list of query strings (e.g., [\"Kyoto to Tokyo train duration\", \"Shinjuku Gyoen opening hours\"]). "
36
+ "If no specific factual research is required, respond with []."
37
+ )
38
+ messages = [
39
+ Message(role="system", content=system_instruction),
40
+ Message(role="user", content=prompt),
41
+ ]
42
+
43
+ queries = []
44
+ try:
45
+ response = self.llm.generate(messages)
46
+ parsed = parse_json_markdown(response.text)
47
+ if isinstance(parsed, list):
48
+ queries = [str(q) for q in parsed]
49
+ except Exception:
50
+ pass
51
+
52
+ return queries
@@ -0,0 +1,14 @@
1
+ """System prompt instructions for the Research Agent."""
2
+
3
+ RESEARCH_SYSTEM_PROMPT = (
4
+ "You are a professional travel research assistant.\n"
5
+ "Your primary goal is to answer travel research questions as accurately, factually, "
6
+ "and realistically as possible based on historical data and travel knowledge.\n"
7
+ "Focus on answering questions about:\n"
8
+ "1. Transit options, durations, routes, and connections (e.g. Kyoto to Hiroshima trains).\n"
9
+ "2. Lodging prices, hostel locations, and standard booking policies.\n"
10
+ "3. Attraction details, ticket prices, opening hours, and scheduled closures.\n"
11
+ "4. Weather patterns, seasonal advice, and local safety updates.\n\n"
12
+ "Be precise and cite standard sources or estimates where appropriate. Keep your response "
13
+ "concise and focused on answering the query factually."
14
+ )
@@ -0,0 +1,5 @@
1
+ """Support-domain agents."""
2
+
3
+ from .triage_agent import SupportTriageAgent
4
+
5
+ __all__ = ["SupportTriageAgent"]