evalrun 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agents/__init__.py +6 -0
- agents/auditor/__init__.py +13 -0
- agents/auditor/budget_auditor.py +91 -0
- agents/auditor/parser.py +139 -0
- agents/auditor/prompts.py +64 -0
- agents/auditor/schema.py +67 -0
- agents/base.py +20 -0
- agents/reflection/__init__.py +3 -0
- agents/reflection/agent.py +77 -0
- agents/reflection/prompts.py +18 -0
- agents/research/__init__.py +4 -0
- agents/research/agent.py +45 -0
- agents/research/planner.py +52 -0
- agents/research/prompts.py +14 -0
- agents/support/__init__.py +5 -0
- agents/support/triage_agent.py +45 -0
- agents/travel/__init__.py +11 -0
- agents/travel/agent.py +377 -0
- agents/travel/prompts.py +30 -0
- agents/travel/session.py +110 -0
- cli/__init__.py +6 -0
- cli/demo.py +47 -0
- cli/formatter.py +93 -0
- cli/html_reporter.py +647 -0
- cli/main.py +423 -0
- cli/progress.py +38 -0
- cli/resolver.py +99 -0
- evalrun-0.4.0.dist-info/METADATA +268 -0
- evalrun-0.4.0.dist-info/RECORD +100 -0
- evalrun-0.4.0.dist-info/WHEEL +5 -0
- evalrun-0.4.0.dist-info/entry_points.txt +2 -0
- evalrun-0.4.0.dist-info/licenses/LICENSE +0 -0
- evalrun-0.4.0.dist-info/top_level.txt +4 -0
- framework/__init__.py +70 -0
- framework/core/__init__.py +17 -0
- framework/core/adapters.py +118 -0
- framework/core/contracts.py +88 -0
- framework/core/suite.py +44 -0
- framework/evaluation/__init__.py +22 -0
- framework/evaluation/base.py +29 -0
- framework/evaluation/dimensions.py +7 -0
- framework/evaluation/engine.py +110 -0
- framework/evaluation/evaluators/__init__.py +7 -0
- framework/evaluation/evaluators/adaptability.py +27 -0
- framework/evaluation/evaluators/base_llm.py +104 -0
- framework/evaluation/evaluators/constraint.py +27 -0
- framework/evaluation/evaluators/information_accuracy.py +41 -0
- framework/evaluation/evaluators/personalization.py +27 -0
- framework/evaluation/evaluators/planning.py +27 -0
- framework/evaluation/evaluators/support.py +81 -0
- framework/evaluation/prompts/__init__.py +11 -0
- framework/evaluation/prompts/adaptability.py +57 -0
- framework/evaluation/prompts/base.py +52 -0
- framework/evaluation/prompts/constraint.py +41 -0
- framework/evaluation/prompts/information_accuracy.py +79 -0
- framework/evaluation/prompts/personalization.py +57 -0
- framework/evaluation/prompts/planning.py +61 -0
- framework/evaluation/runner.py +363 -0
- framework/evaluation/testing.py +25 -0
- framework/exceptions.py +49 -0
- framework/llms/__init__.py +8 -0
- framework/llms/base.py +37 -0
- framework/llms/factory.py +38 -0
- framework/llms/gemini.py +85 -0
- framework/llms/mock.py +25 -0
- framework/llms/openai.py +94 -0
- framework/llms/openai_compatible.py +139 -0
- framework/mcp/__init__.py +20 -0
- framework/mcp/client.py +62 -0
- framework/mcp/constraints.py +125 -0
- framework/mcp/revision_summary.py +122 -0
- framework/mcp/server.py +49 -0
- framework/memory/__init__.py +3 -0
- framework/memory/base.py +17 -0
- framework/models.py +83 -0
- framework/parser.py +45 -0
- framework/parsers/__init__.py +12 -0
- framework/parsers/frontmatter.py +28 -0
- framework/parsers/mapper.py +66 -0
- framework/parsers/markdown.py +122 -0
- framework/parsers/transformers.py +112 -0
- framework/profiles/__init__.py +28 -0
- framework/profiles/registry.py +104 -0
- framework/profiles/support.py +27 -0
- framework/profiles/travel.py +78 -0
- framework/regression/__init__.py +17 -0
- framework/regression/comparator.py +273 -0
- framework/regression/loader.py +145 -0
- framework/sdk.py +151 -0
- framework/utils.py +41 -0
- framework/verification/__init__.py +13 -0
- framework/verification/base.py +32 -0
- framework/verification/extractor.py +105 -0
- framework/verification/local.py +122 -0
- framework/verification/models.py +112 -0
- framework/verification/pipeline.py +36 -0
- framework/verification/prompts.py +24 -0
- framework/verification/utils.py +54 -0
- ui/__init__.py +1 -0
- ui/server.py +255 -0
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""Prompt builder for Adaptability dimension."""
|
|
2
|
+
|
|
3
|
+
from typing import List
|
|
4
|
+
from framework.models import AgentOutput, Benchmark
|
|
5
|
+
from framework.evaluation.prompts.base import build_llm_judge_prompt
|
|
6
|
+
from framework.llms import Message
|
|
7
|
+
|
|
8
|
+
# Detailed evaluation rubric for Adaptability
|
|
9
|
+
ADAPTABILITY_RUBRIC = (
|
|
10
|
+
"Main Question: When circumstances change, how well does the agent revise "
|
|
11
|
+
"the plan while preserving the overall experience?\n\n"
|
|
12
|
+
"Note on scope: Evaluate only how well the agent adapts the itinerary to changes, "
|
|
13
|
+
"disruptions, or new constraints introduced mid-travel. Do not evaluate baseline "
|
|
14
|
+
"planning quality, factual correctness of attractions, or static user interests, "
|
|
15
|
+
"as those are evaluated separately in other dimensions.\n\n"
|
|
16
|
+
"Evaluate the adaptability quality of the travel itinerary across these three core aspects:\n\n"
|
|
17
|
+
"1. Constraint Adaptation:\n"
|
|
18
|
+
" Assess how effectively the agent revises the itinerary to accommodate modified or new constraints.\n"
|
|
19
|
+
" Favor revisions that make the smallest necessary changes while preserving the overall itinerary whenever practical.\n"
|
|
20
|
+
" Consider factors such as: adapting to budget reductions, timeline shifts, or transit cancellations.\n\n"
|
|
21
|
+
"2. Experience Preservation:\n"
|
|
22
|
+
" Assess how well the revised plan preserves the original travel goals, style, and interests of the traveler.\n"
|
|
23
|
+
" Consider factors such as: finding alternative activities of similar theme/vibe rather than replacing them with generic tourist options.\n\n"
|
|
24
|
+
"3. Impact Mitigation:\n"
|
|
25
|
+
" Assess whether the revisions minimize cascading disruptions to the rest of the itinerary.\n"
|
|
26
|
+
" Consider factors such as: avoiding unnecessary cancellations of existing bookings and keeping alterations localized where possible.\n\n"
|
|
27
|
+
"### Scoring Guidance:\n"
|
|
28
|
+
"- 90–100: Excellent adaptability, smoothly resolving changes with zero friction and high experience preservation.\n"
|
|
29
|
+
"- 75–89: Good adaptability resolving disruptions with minor inefficiencies or pacing compromises.\n"
|
|
30
|
+
"- 50–74: Moderate adaptability; recovers from changes but sacrifices multiple travel goals or introduces excessive pacing stress.\n"
|
|
31
|
+
"- 25–49: Poor adaptability; fails to meet new constraints or completely loses the traveler's preferences during revision.\n"
|
|
32
|
+
"- 0–24: Not adaptable; fails to address changed circumstances entirely.\n\n"
|
|
33
|
+
"Assign the score holistically across all three aspects. "
|
|
34
|
+
"Do not average the aspects mechanically unless the evidence clearly supports doing so."
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def build_adaptability_prompt(
|
|
39
|
+
benchmark: Benchmark, output: AgentOutput
|
|
40
|
+
) -> List[Message]:
|
|
41
|
+
"""Generates LLM judge messages for adaptability quality assessment.
|
|
42
|
+
|
|
43
|
+
Args:
|
|
44
|
+
benchmark: The scenario benchmark details.
|
|
45
|
+
output: The agent response output.
|
|
46
|
+
|
|
47
|
+
Returns:
|
|
48
|
+
List of prompt Message structures.
|
|
49
|
+
"""
|
|
50
|
+
system_objective = (
|
|
51
|
+
"You are an objective AI evaluator checking the adaptability quality "
|
|
52
|
+
"of an agent's output itinerary against a benchmark scenario."
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
return build_llm_judge_prompt(
|
|
56
|
+
system_objective, ADAPTABILITY_RUBRIC, benchmark, output
|
|
57
|
+
)
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
"""Base template components and helpers for building evaluation prompts."""
|
|
2
|
+
|
|
3
|
+
from typing import List
|
|
4
|
+
from framework.models import AgentOutput, Benchmark
|
|
5
|
+
from framework.llms import Message
|
|
6
|
+
|
|
7
|
+
# Standard JSON schema instructions enforced across all LLM judges
|
|
8
|
+
JSON_RESPONSE_SCHEMA = (
|
|
9
|
+
"Return a JSON object containing exactly these fields:\n"
|
|
10
|
+
"{\n"
|
|
11
|
+
' "score": <integer from 0 to 100 representing the grade>,\n'
|
|
12
|
+
' "reason": "<detailed justification detailing your assessment>"\n'
|
|
13
|
+
"}\n"
|
|
14
|
+
"Do not include any markdown syntax, explanations, or code blocks outside the JSON."
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def build_llm_judge_prompt(
|
|
19
|
+
system_objective: str,
|
|
20
|
+
criteria_rubric: str,
|
|
21
|
+
benchmark: Benchmark,
|
|
22
|
+
output: AgentOutput,
|
|
23
|
+
) -> List[Message]:
|
|
24
|
+
"""Universal prompt builder for LLM judges.
|
|
25
|
+
|
|
26
|
+
Ensures consistent formatting and JSON output constraints.
|
|
27
|
+
|
|
28
|
+
Args:
|
|
29
|
+
system_objective: The role and primary mission of the judge.
|
|
30
|
+
criteria_rubric: Rubric details specific to the evaluation dimension.
|
|
31
|
+
benchmark: The benchmark scenario details.
|
|
32
|
+
output: The agent response output.
|
|
33
|
+
|
|
34
|
+
Returns:
|
|
35
|
+
A list of chat Message structures.
|
|
36
|
+
"""
|
|
37
|
+
system_instruction = (
|
|
38
|
+
f"{system_objective}\n\n"
|
|
39
|
+
f"### Rubric / Criteria:\n{criteria_rubric}\n\n"
|
|
40
|
+
f"{JSON_RESPONSE_SCHEMA}"
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
user_prompt = (
|
|
44
|
+
f"Benchmark Name: {benchmark.name}\n"
|
|
45
|
+
f"Benchmark Description: {benchmark.description}\n\n"
|
|
46
|
+
f"### Agent Output To Evaluate:\n{output.content}\n"
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
return [
|
|
50
|
+
Message(role="system", content=system_instruction),
|
|
51
|
+
Message(role="user", content=user_prompt),
|
|
52
|
+
]
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Prompt builder for Constraint Satisfaction dimension."""
|
|
2
|
+
|
|
3
|
+
from typing import List
|
|
4
|
+
import yaml
|
|
5
|
+
from framework.models import AgentOutput, Benchmark
|
|
6
|
+
from framework.evaluation.prompts.base import build_llm_judge_prompt
|
|
7
|
+
from framework.llms import Message
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def build_constraint_prompt(
|
|
11
|
+
benchmark: Benchmark, output: AgentOutput
|
|
12
|
+
) -> List[Message]:
|
|
13
|
+
"""Generates LLM judge messages for constraint satisfaction checking.
|
|
14
|
+
|
|
15
|
+
Args:
|
|
16
|
+
benchmark: The scenario benchmark details.
|
|
17
|
+
output: The agent response output.
|
|
18
|
+
|
|
19
|
+
Returns:
|
|
20
|
+
List of prompt Message structures.
|
|
21
|
+
"""
|
|
22
|
+
system_objective = (
|
|
23
|
+
"You are an objective AI evaluator checking if an agent's output satisfies "
|
|
24
|
+
"the explicit constraints of a benchmark scenario."
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
constraints_str = (
|
|
28
|
+
yaml.dump(benchmark.constraints, default_flow_style=False)
|
|
29
|
+
if benchmark.constraints
|
|
30
|
+
else "No explicit constraints defined."
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
criteria_rubric = (
|
|
34
|
+
"Analyze the agent output and determine if every constraint listed "
|
|
35
|
+
"below is satisfied.\n"
|
|
36
|
+
f"Explicit constraints to verify:\n{constraints_str}"
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
return build_llm_judge_prompt(
|
|
40
|
+
system_objective, criteria_rubric, benchmark, output
|
|
41
|
+
)
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
"""Prompt builder for Information Accuracy dimension."""
|
|
2
|
+
|
|
3
|
+
from typing import List
|
|
4
|
+
from framework.models import AgentOutput, Benchmark
|
|
5
|
+
from framework.evaluation.prompts.base import JSON_RESPONSE_SCHEMA
|
|
6
|
+
from framework.llms import Message
|
|
7
|
+
from framework.verification.models import VerificationReport
|
|
8
|
+
|
|
9
|
+
# Detailed evaluation rubric for Information Accuracy
|
|
10
|
+
INFORMATION_ACCURACY_RUBRIC = (
|
|
11
|
+
"Main Question: Considering only information accuracy, can the factual claims "
|
|
12
|
+
"in this itinerary be trusted?\n\n"
|
|
13
|
+
"Note on scope: Evaluate only the truthfulness, precision, and validity of factual claims "
|
|
14
|
+
"(e.g., hotel prices, attraction existence, opening hours, transit times) based on the provided "
|
|
15
|
+
"verification evidence. Do not evaluate planning quality, constraint satisfaction, personalization, "
|
|
16
|
+
"or adaptability, as those are evaluated separately in other dimensions.\n\n"
|
|
17
|
+
"Evaluate the information accuracy of the travel itinerary based on this verification evidence across three core aspects:\n\n"
|
|
18
|
+
"1. Verification Accuracy:\n"
|
|
19
|
+
" Assess the proportion of claims that were verified as true.\n"
|
|
20
|
+
" Consider factors such as: the number of refuted claims, particularly those that are critical to the trip.\n\n"
|
|
21
|
+
"2. Severity of Refutations:\n"
|
|
22
|
+
" Assess the real-world impact of incorrect claims on the traveler's experience.\n"
|
|
23
|
+
" Treat critical factual inaccuracies—such as incorrect visa/entry rules, non-existent attractions, "
|
|
24
|
+
" closed attractions, or severe routing/geographical errors—as significantly more critical "
|
|
25
|
+
" than minor discrepancies (like small price or budget differences) when assessing the scoring deductions.\n\n"
|
|
26
|
+
"3. Handling of Unverifiable or Missing Facts:\n"
|
|
27
|
+
" Assess how the verifier findings (unknown or not found claims) impact the overall trustworthiness of the itinerary.\n\n"
|
|
28
|
+
"### Scoring Guidance:\n"
|
|
29
|
+
"- 90–100: Highly accurate. All claims are verified, or only minor, trivial discrepancies exist.\n"
|
|
30
|
+
"- 75–89: Generally accurate. Minor price or timing errors that do not disrupt the trip flow.\n"
|
|
31
|
+
"- 50–74: Noticeable factual errors. Multiple refuted claims or a single critical refutation (e.g., closed attraction).\n"
|
|
32
|
+
"- 25–49: Major inaccuracies. Key transport links or multiple attractions are closed/incorrect.\n"
|
|
33
|
+
"- 0–24: Highly untrustworthy. Multiple severe refutations, non-existent locations, or major fictional data.\n\n"
|
|
34
|
+
"Assign the score holistically across all three aspects. "
|
|
35
|
+
"Do not average the aspects mechanically unless the evidence clearly supports doing so."
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def build_information_accuracy_prompt(
|
|
40
|
+
benchmark: Benchmark,
|
|
41
|
+
output: AgentOutput,
|
|
42
|
+
report: VerificationReport,
|
|
43
|
+
) -> List[Message]:
|
|
44
|
+
"""Generates LLM judge messages for information accuracy assessment.
|
|
45
|
+
|
|
46
|
+
Args:
|
|
47
|
+
benchmark: The scenario benchmark details.
|
|
48
|
+
output: The agent response output.
|
|
49
|
+
report: The VerificationReport containing factual verification results.
|
|
50
|
+
|
|
51
|
+
Returns:
|
|
52
|
+
List of prompt Message structures.
|
|
53
|
+
"""
|
|
54
|
+
system_instruction = (
|
|
55
|
+
"You are an objective AI evaluator checking the information accuracy "
|
|
56
|
+
"of an agent's output itinerary against a benchmark scenario and "
|
|
57
|
+
"provided verification evidence.\n\n"
|
|
58
|
+
f"### Rubric / Criteria:\n{INFORMATION_ACCURACY_RUBRIC}\n\n"
|
|
59
|
+
f"{JSON_RESPONSE_SCHEMA}"
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
evidence_context = report.to_markdown()
|
|
63
|
+
|
|
64
|
+
user_prompt = (
|
|
65
|
+
f"VERIFICATION EVIDENCE FINDINGS:\n"
|
|
66
|
+
f"================================\n"
|
|
67
|
+
f"{evidence_context}\n"
|
|
68
|
+
f"================================\n\n"
|
|
69
|
+
f"Please evaluate the original itinerary below considering this factual evidence.\n\n"
|
|
70
|
+
f"Benchmark Name: {benchmark.name}\n"
|
|
71
|
+
f"Benchmark Description: {benchmark.description}\n\n"
|
|
72
|
+
f"### Agent Output To Evaluate:\n{output.content}\n"
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
return [
|
|
76
|
+
Message(role="system", content=system_instruction),
|
|
77
|
+
Message(role="user", content=user_prompt),
|
|
78
|
+
]
|
|
79
|
+
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""Prompt builder for Personalization dimension."""
|
|
2
|
+
|
|
3
|
+
from typing import List
|
|
4
|
+
from framework.models import AgentOutput, Benchmark
|
|
5
|
+
from framework.evaluation.prompts.base import build_llm_judge_prompt
|
|
6
|
+
from framework.llms import Message
|
|
7
|
+
|
|
8
|
+
# Detailed evaluation rubric for Personalization
|
|
9
|
+
PERSONALIZATION_RUBRIC = (
|
|
10
|
+
"Main Question: Considering only personalization, does the itinerary feel "
|
|
11
|
+
"specifically tailored to this traveler's stated preferences, interests, and profile?\n\n"
|
|
12
|
+
"Note on scope: Evaluate only how well the itinerary aligns with the traveler's stated preferences, "
|
|
13
|
+
"goals, interests, and style. Do not evaluate route efficiency, daily pacing, factual correctness "
|
|
14
|
+
"of attractions, or strict duration constraints, as those are evaluated separately in other dimensions.\n\n"
|
|
15
|
+
"Evaluate the personalization quality of the travel itinerary across these three core aspects:\n\n"
|
|
16
|
+
"1. Preference Alignment:\n"
|
|
17
|
+
" Assess whether the activities, sightseeing spots, and recommendations match the traveler's listed interests.\n"
|
|
18
|
+
" Consider both explicit preferences stated by the traveler and reasonable implications of those preferences "
|
|
19
|
+
" (e.g., backpacking implies budget lodging and public transit), but do not invent unsupported interests.\n"
|
|
20
|
+
" Consider factors such as: incorporating stated hobbies (e.g., photography, thrift shopping, café culture) and aligning with the backpacking travel style.\n\n"
|
|
21
|
+
"2. Context & Constraints Integration:\n"
|
|
22
|
+
" Assess whether lodging selection and pacing integrate cleanly with the traveler's daily constraints.\n"
|
|
23
|
+
" Consider factors such as: scheduling around remote work windows and respecting walking preferences or physical limits.\n\n"
|
|
24
|
+
"3. Recommendation Relevance:\n"
|
|
25
|
+
" Assess whether the recommendations are highly relevant to this specific traveler's profile rather than boilerplate tourist stops.\n"
|
|
26
|
+
" Consider factors such as: localized recommendations that directly serve the traveler's stated and implied interests.\n\n"
|
|
27
|
+
"### Scoring Guidance:\n"
|
|
28
|
+
"- 90–100: Excellent personalization, highly tailored to the traveler's profile with deep alignment.\n"
|
|
29
|
+
"- 75–89: Good personalization matching major interests, with minor generic elements.\n"
|
|
30
|
+
"- 50–74: Moderate personalization; feels somewhat generic or overlooks multiple stated preferences.\n"
|
|
31
|
+
"- 25–49: Poor personalization; ignores key interests or remote work requirements.\n"
|
|
32
|
+
"- 0–24: Not personalized at all; entirely boilerplate itinerary.\n\n"
|
|
33
|
+
"Assign the score holistically across all three aspects. "
|
|
34
|
+
"Do not average the aspects mechanically unless the evidence clearly supports doing so."
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def build_personalization_prompt(
|
|
39
|
+
benchmark: Benchmark, output: AgentOutput
|
|
40
|
+
) -> List[Message]:
|
|
41
|
+
"""Generates LLM judge messages for personalization quality assessment.
|
|
42
|
+
|
|
43
|
+
Args:
|
|
44
|
+
benchmark: The scenario benchmark details.
|
|
45
|
+
output: The agent response output.
|
|
46
|
+
|
|
47
|
+
Returns:
|
|
48
|
+
List of prompt Message structures.
|
|
49
|
+
"""
|
|
50
|
+
system_objective = (
|
|
51
|
+
"You are an objective AI evaluator checking the personalization quality "
|
|
52
|
+
"of an agent's output itinerary against a benchmark scenario."
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
return build_llm_judge_prompt(
|
|
56
|
+
system_objective, PERSONALIZATION_RUBRIC, benchmark, output
|
|
57
|
+
)
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""Prompt builder for Planning Quality dimension."""
|
|
2
|
+
|
|
3
|
+
from typing import List
|
|
4
|
+
from framework.models import AgentOutput, Benchmark
|
|
5
|
+
from framework.evaluation.prompts.base import build_llm_judge_prompt
|
|
6
|
+
from framework.llms import Message
|
|
7
|
+
|
|
8
|
+
# Detailed evaluation rubric for Planning Quality
|
|
9
|
+
PLANNING_RUBRIC = (
|
|
10
|
+
"Main Question: Considering only planning quality, is this itinerary organized in the best possible way for the traveler?\n\n"
|
|
11
|
+
"Note on scope: Evaluate only the organization and structure of the itinerary. "
|
|
12
|
+
"Do not consider factual accuracy, personalization, or whether explicit user constraints were satisfied, "
|
|
13
|
+
"as those are evaluated separately in other dimensions.\n\n"
|
|
14
|
+
"Evaluate the planning quality of the travel itinerary across these six core aspects:\n\n"
|
|
15
|
+
"1. Route Logic & Geographic Efficiency:\n"
|
|
16
|
+
" Assess whether the itinerary minimizes unnecessary travel while maintaining a logical progression.\n"
|
|
17
|
+
" Consider factors such as: unnecessary backtracking, revisiting cities, or inefficient long-distance jumps.\n\n"
|
|
18
|
+
"2. Accommodation Stability:\n"
|
|
19
|
+
" Assess whether stays are grouped effectively to avoid unnecessary overhead.\n"
|
|
20
|
+
" Consider factors such as: changing hotels/guesthouses every night, daily packing/unpacking, or not grouping activities around the active lodging node.\n\n"
|
|
21
|
+
"3. Daily Pacing & Balance:\n"
|
|
22
|
+
" Assess whether the activities are realistic and pacing feels balanced.\n"
|
|
23
|
+
" Consider factors such as: overloading single days with 5+ major sites, not leaving transit buffers, or conflicting with remote work hours.\n\n"
|
|
24
|
+
"4. Time & Transit Utilization:\n"
|
|
25
|
+
" Assess whether transit schedules are optimized for convenience and cost.\n"
|
|
26
|
+
" Consider factors such as: losing full days to slow daylight transits, choosing complex routes with tight/unrealistic transfer connections.\n\n"
|
|
27
|
+
"5. Trip Flow & Cohesion:\n"
|
|
28
|
+
" Assess whether the sequence of cities, theme, and travel flow feel natural and coherent rather than disjointed, scattered, or random.\n\n"
|
|
29
|
+
"6. Trade-off Optimization:\n"
|
|
30
|
+
" Assess whether the plan demonstrates balanced decisions between travel convenience, fatigue, cost, and traveler experience.\n\n"
|
|
31
|
+
"### Scoring Guidance:\n"
|
|
32
|
+
"- 90–100: Excellent planning with only minor improvements possible.\n"
|
|
33
|
+
"- 75–89: Good planning with some inefficiencies.\n"
|
|
34
|
+
"- 50–74: Noticeable planning problems that reduce the overall experience.\n"
|
|
35
|
+
"- 25–49: Major inefficiencies or unrealistic daily pacing.\n"
|
|
36
|
+
"- 0–24: Poorly organized itinerary.\n\n"
|
|
37
|
+
"Assign the score holistically across all six aspects. "
|
|
38
|
+
"Do not average the aspects mechanically unless the evidence clearly supports doing so."
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def build_planning_prompt(
|
|
43
|
+
benchmark: Benchmark, output: AgentOutput
|
|
44
|
+
) -> List[Message]:
|
|
45
|
+
"""Generates LLM judge messages for planning quality assessment.
|
|
46
|
+
|
|
47
|
+
Args:
|
|
48
|
+
benchmark: The scenario benchmark details.
|
|
49
|
+
output: The agent response output.
|
|
50
|
+
|
|
51
|
+
Returns:
|
|
52
|
+
List of prompt Message structures.
|
|
53
|
+
"""
|
|
54
|
+
system_objective = (
|
|
55
|
+
"You are an objective AI evaluator checking the planning quality "
|
|
56
|
+
"of an agent's output itinerary against a benchmark scenario."
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
return build_llm_judge_prompt(
|
|
60
|
+
system_objective, PLANNING_RUBRIC, benchmark, output
|
|
61
|
+
)
|