evalrun 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. agents/__init__.py +6 -0
  2. agents/auditor/__init__.py +13 -0
  3. agents/auditor/budget_auditor.py +91 -0
  4. agents/auditor/parser.py +139 -0
  5. agents/auditor/prompts.py +64 -0
  6. agents/auditor/schema.py +67 -0
  7. agents/base.py +20 -0
  8. agents/reflection/__init__.py +3 -0
  9. agents/reflection/agent.py +77 -0
  10. agents/reflection/prompts.py +18 -0
  11. agents/research/__init__.py +4 -0
  12. agents/research/agent.py +45 -0
  13. agents/research/planner.py +52 -0
  14. agents/research/prompts.py +14 -0
  15. agents/support/__init__.py +5 -0
  16. agents/support/triage_agent.py +45 -0
  17. agents/travel/__init__.py +11 -0
  18. agents/travel/agent.py +377 -0
  19. agents/travel/prompts.py +30 -0
  20. agents/travel/session.py +110 -0
  21. cli/__init__.py +6 -0
  22. cli/demo.py +47 -0
  23. cli/formatter.py +93 -0
  24. cli/html_reporter.py +647 -0
  25. cli/main.py +423 -0
  26. cli/progress.py +38 -0
  27. cli/resolver.py +99 -0
  28. evalrun-0.4.0.dist-info/METADATA +268 -0
  29. evalrun-0.4.0.dist-info/RECORD +100 -0
  30. evalrun-0.4.0.dist-info/WHEEL +5 -0
  31. evalrun-0.4.0.dist-info/entry_points.txt +2 -0
  32. evalrun-0.4.0.dist-info/licenses/LICENSE +0 -0
  33. evalrun-0.4.0.dist-info/top_level.txt +4 -0
  34. framework/__init__.py +70 -0
  35. framework/core/__init__.py +17 -0
  36. framework/core/adapters.py +118 -0
  37. framework/core/contracts.py +88 -0
  38. framework/core/suite.py +44 -0
  39. framework/evaluation/__init__.py +22 -0
  40. framework/evaluation/base.py +29 -0
  41. framework/evaluation/dimensions.py +7 -0
  42. framework/evaluation/engine.py +110 -0
  43. framework/evaluation/evaluators/__init__.py +7 -0
  44. framework/evaluation/evaluators/adaptability.py +27 -0
  45. framework/evaluation/evaluators/base_llm.py +104 -0
  46. framework/evaluation/evaluators/constraint.py +27 -0
  47. framework/evaluation/evaluators/information_accuracy.py +41 -0
  48. framework/evaluation/evaluators/personalization.py +27 -0
  49. framework/evaluation/evaluators/planning.py +27 -0
  50. framework/evaluation/evaluators/support.py +81 -0
  51. framework/evaluation/prompts/__init__.py +11 -0
  52. framework/evaluation/prompts/adaptability.py +57 -0
  53. framework/evaluation/prompts/base.py +52 -0
  54. framework/evaluation/prompts/constraint.py +41 -0
  55. framework/evaluation/prompts/information_accuracy.py +79 -0
  56. framework/evaluation/prompts/personalization.py +57 -0
  57. framework/evaluation/prompts/planning.py +61 -0
  58. framework/evaluation/runner.py +363 -0
  59. framework/evaluation/testing.py +25 -0
  60. framework/exceptions.py +49 -0
  61. framework/llms/__init__.py +8 -0
  62. framework/llms/base.py +37 -0
  63. framework/llms/factory.py +38 -0
  64. framework/llms/gemini.py +85 -0
  65. framework/llms/mock.py +25 -0
  66. framework/llms/openai.py +94 -0
  67. framework/llms/openai_compatible.py +139 -0
  68. framework/mcp/__init__.py +20 -0
  69. framework/mcp/client.py +62 -0
  70. framework/mcp/constraints.py +125 -0
  71. framework/mcp/revision_summary.py +122 -0
  72. framework/mcp/server.py +49 -0
  73. framework/memory/__init__.py +3 -0
  74. framework/memory/base.py +17 -0
  75. framework/models.py +83 -0
  76. framework/parser.py +45 -0
  77. framework/parsers/__init__.py +12 -0
  78. framework/parsers/frontmatter.py +28 -0
  79. framework/parsers/mapper.py +66 -0
  80. framework/parsers/markdown.py +122 -0
  81. framework/parsers/transformers.py +112 -0
  82. framework/profiles/__init__.py +28 -0
  83. framework/profiles/registry.py +104 -0
  84. framework/profiles/support.py +27 -0
  85. framework/profiles/travel.py +78 -0
  86. framework/regression/__init__.py +17 -0
  87. framework/regression/comparator.py +273 -0
  88. framework/regression/loader.py +145 -0
  89. framework/sdk.py +151 -0
  90. framework/utils.py +41 -0
  91. framework/verification/__init__.py +13 -0
  92. framework/verification/base.py +32 -0
  93. framework/verification/extractor.py +105 -0
  94. framework/verification/local.py +122 -0
  95. framework/verification/models.py +112 -0
  96. framework/verification/pipeline.py +36 -0
  97. framework/verification/prompts.py +24 -0
  98. framework/verification/utils.py +54 -0
  99. ui/__init__.py +1 -0
  100. ui/server.py +255 -0
@@ -0,0 +1,88 @@
1
+ """Domain-neutral core contracts and data models for evaluation."""
2
+
3
+ from dataclasses import dataclass, field
4
+ from datetime import datetime, timezone
5
+ from typing import Any, Dict, List, Literal, Optional
6
+ from framework.models import Benchmark
7
+
8
+
9
+ @dataclass
10
+ class Scenario:
11
+ """Domain-neutral evaluation scenario specification."""
12
+
13
+ id: str
14
+ name: str
15
+ domain: str # e.g. "travel", "support_triage", "scheduling"
16
+ description: str
17
+ prompt: str
18
+ constraints: Dict[str, Any]
19
+ expected_behavior: List[str]
20
+ evaluation_criteria: Dict[str, List[str]]
21
+ pass_criteria: List[str]
22
+ failure_conditions: List[str]
23
+ notes: Optional[List[str]] = None
24
+ profile_name: str = "travel-agent"
25
+ metadata: Dict[str, Any] = field(default_factory=dict)
26
+
27
+ @property
28
+ def benchmark_id(self) -> str:
29
+ """Backward-compatibility property for legacy codebase."""
30
+ return self.id
31
+
32
+ @property
33
+ def profile(self) -> str:
34
+ """Backward-compatibility property for legacy profile name."""
35
+ return self.profile_name
36
+
37
+
38
+ def to_scenario(benchmark: Benchmark, domain: str = "travel") -> Scenario:
39
+ """Converts a legacy Benchmark instance into a generic Scenario instance."""
40
+ return Scenario(
41
+ id=benchmark.benchmark_id,
42
+ name=benchmark.name,
43
+ domain=domain,
44
+ description=benchmark.description,
45
+ prompt=benchmark.prompt,
46
+ constraints=benchmark.constraints,
47
+ expected_behavior=benchmark.expected_behavior,
48
+ evaluation_criteria=benchmark.evaluation_criteria,
49
+ pass_criteria=benchmark.pass_criteria,
50
+ failure_conditions=benchmark.failure_conditions,
51
+ notes=benchmark.notes,
52
+ profile_name=getattr(benchmark, "profile", "travel-agent"),
53
+ )
54
+
55
+
56
+ def to_benchmark(scenario: Scenario) -> Benchmark:
57
+ """Converts a generic Scenario instance into a legacy Benchmark instance."""
58
+ return Benchmark(
59
+ benchmark_id=scenario.id,
60
+ name=scenario.name,
61
+ description=scenario.description,
62
+ prompt=scenario.prompt,
63
+ constraints=scenario.constraints,
64
+ expected_behavior=scenario.expected_behavior,
65
+ evaluation_criteria=scenario.evaluation_criteria,
66
+ pass_criteria=scenario.pass_criteria,
67
+ failure_conditions=scenario.failure_conditions,
68
+ notes=scenario.notes or [],
69
+ profile=scenario.profile_name,
70
+ )
71
+
72
+
73
+ @dataclass
74
+ class RunTrace:
75
+ """Execution trace tracking metadata, latency, token usage, and status for an evaluation run."""
76
+
77
+ trace_id: str
78
+ scenario_id: str
79
+ agent_id: str
80
+ model_name: str
81
+ started_at_utc: str
82
+ finished_at_utc: str
83
+ latency_seconds: float
84
+ status: Literal["success", "error", "timeout"]
85
+ error: Optional[str] = None
86
+ retries_attempted: int = 0
87
+ token_usage: Optional[Dict[str, int]] = None
88
+ metadata: Dict[str, Any] = field(default_factory=dict)
@@ -0,0 +1,44 @@
1
+ """EvaluationSuite class managing versioned collections of Scenarios and EvaluationProfiles."""
2
+
3
+ from dataclasses import dataclass, field
4
+ from typing import Dict, List, Optional
5
+
6
+ from framework.core.contracts import Scenario
7
+ from framework.models import EvaluationProfile
8
+
9
+
10
+ @dataclass
11
+ class EvaluationSuite:
12
+ """Encapsulates a collection of scenarios and domain evaluation profiles."""
13
+
14
+ suite_id: str
15
+ name: str
16
+ domain: str
17
+ version: str
18
+ scenarios: List[Scenario]
19
+ profiles: Dict[str, EvaluationProfile] = field(default_factory=dict)
20
+
21
+ def get_scenario(self, scenario_id: str) -> Optional[Scenario]:
22
+ """Looks up a scenario by ID."""
23
+ for scenario in self.scenarios:
24
+ if scenario.id == scenario_id:
25
+ return scenario
26
+ return None
27
+
28
+ def get_profile(self, profile_name: Optional[str] = None) -> EvaluationProfile:
29
+ """Resolves an evaluation profile by name or default key."""
30
+ if profile_name:
31
+ if profile_name in self.profiles:
32
+ return self.profiles[profile_name]
33
+ available = list(self.profiles.keys())
34
+ raise ValueError(
35
+ f"Profile '{profile_name}' not found in suite '{self.suite_id}'. "
36
+ f"Available profiles: {available}"
37
+ )
38
+ if "default" in self.profiles:
39
+ return self.profiles["default"]
40
+ available = list(self.profiles.keys())
41
+ raise ValueError(
42
+ f"No profile specified and no 'default' profile registered for suite '{self.suite_id}'. "
43
+ f"Available profiles: {available}"
44
+ )
@@ -0,0 +1,22 @@
1
+ """Package containing the evaluation engine, interfaces, and concrete evaluators."""
2
+
3
+ from framework.evaluation.base import BaseEvaluator
4
+ from framework.evaluation.engine import EvaluationEngine
5
+ from framework.evaluation.runner import BenchmarkRunner
6
+ from framework.llms import BaseLLM, MockLLM, OpenAILLM, GeminiLLM
7
+ from framework.evaluation.evaluators import (
8
+ ConstraintEvaluator,
9
+ PlanningQualityEvaluator,
10
+ PersonalizationEvaluator,
11
+ AdaptabilityEvaluator,
12
+ InformationAccuracyEvaluator,
13
+ )
14
+ from framework.evaluation.evaluators.base_llm import BaseLLMEvaluator
15
+ from framework.evaluation.testing import DummyEvaluator
16
+ from framework.evaluation.dimensions import (
17
+ CONSTRAINT_SATISFACTION,
18
+ PLANNING_QUALITY,
19
+ INFORMATION_ACCURACY,
20
+ PERSONALIZATION,
21
+ ADAPTABILITY,
22
+ )
@@ -0,0 +1,29 @@
1
+ """Base interface for all evaluation strategies."""
2
+
3
+ from abc import ABC, abstractmethod
4
+ from framework.models import AgentOutput, Benchmark, DimensionScore
5
+
6
+
7
+ class BaseEvaluator(ABC):
8
+ """Abstract base class defining the interface for evaluation strategies.
9
+
10
+ Evaluators assess an agent's output against a benchmark scenario for a
11
+ single specific evaluation dimension.
12
+ """
13
+
14
+ @abstractmethod
15
+ def evaluate(
16
+ self,
17
+ benchmark: Benchmark,
18
+ output: AgentOutput,
19
+ ) -> DimensionScore:
20
+ """Evaluates a single dimension of an agent's output against a benchmark.
21
+
22
+ Args:
23
+ benchmark: The scenario benchmark details.
24
+ output: The response/data generated by the AI agent.
25
+
26
+ Returns:
27
+ A DimensionScore containing the dimension name, score, and reasoning.
28
+ """
29
+ pass
@@ -0,0 +1,7 @@
1
+ """Standard evaluation dimension constants."""
2
+
3
+ CONSTRAINT_SATISFACTION = "Constraint Satisfaction"
4
+ PLANNING_QUALITY = "Planning Quality"
5
+ INFORMATION_ACCURACY = "Information Accuracy"
6
+ PERSONALIZATION = "Personalization"
7
+ ADAPTABILITY = "Adaptability"
@@ -0,0 +1,110 @@
1
+ from langfuse import observe
2
+
3
+ from typing import Dict
4
+ from framework.exceptions import EvaluationError
5
+ from framework.models import (
6
+ AgentOutput,
7
+ Benchmark,
8
+ DimensionScore,
9
+ EvaluationProfile,
10
+ EvaluationResult,
11
+ )
12
+ from framework.evaluation.base import BaseEvaluator
13
+
14
+
15
+ class EvaluationEngine:
16
+ """Orchestrates the evaluation of an agent's output against a benchmark scenario.
17
+
18
+ Uses an EvaluationProfile to weigh scores across multiple dimensions and
19
+ calculates the final evaluation outcome.
20
+ """
21
+
22
+ def __init__(self, evaluators: Dict[str, BaseEvaluator]):
23
+ """Initializes the EvaluationEngine.
24
+
25
+ Args:
26
+ evaluators: A dictionary mapping dimension names to BaseEvaluator instances.
27
+ """
28
+ self.evaluators = evaluators
29
+
30
+ @observe(name="evaluation-engine")
31
+ def evaluate(
32
+ self,
33
+ benchmark: Benchmark,
34
+ output: AgentOutput,
35
+ profile: EvaluationProfile,
36
+ ) -> EvaluationResult:
37
+ """Runs the evaluation pipeline for the given benchmark and agent output.
38
+
39
+ Args:
40
+ benchmark: The benchmark scenario.
41
+ output: The response/data generated by the AI agent.
42
+ profile: The evaluation profile specifying dimension weights.
43
+
44
+ Returns:
45
+ An EvaluationResult containing the dimension breakdown and final status.
46
+
47
+ Raises:
48
+ EvaluationError: If a dimension specified in the profile has no registered evaluator,
49
+ is missing from the benchmark criteria, or returns an inconsistent dimension name.
50
+ """
51
+ if not profile.weights:
52
+ raise EvaluationError(
53
+ f"Evaluation profile '{profile.name}' contains no weights."
54
+ )
55
+
56
+ # Validate that all profile dimensions exist in the benchmark criteria
57
+ for dimension in profile.weights.keys():
58
+ if dimension not in benchmark.evaluation_criteria:
59
+ raise EvaluationError(
60
+ f"Dimension '{dimension}' from profile '{profile.name}' "
61
+ f"is not defined in benchmark '{benchmark.benchmark_id}' evaluation criteria."
62
+ )
63
+
64
+ dimension_scores = []
65
+ weighted_score_sum = 0.0
66
+ total_weight = 0.0
67
+
68
+ for dimension, weight in profile.weights.items():
69
+ evaluator = self.evaluators.get(dimension)
70
+ if not evaluator:
71
+ raise EvaluationError(
72
+ f"No evaluator registered for dimension '{dimension}' "
73
+ f"in profile '{profile.name}'."
74
+ )
75
+
76
+ try:
77
+ score = evaluator.evaluate(benchmark, output)
78
+ except Exception as e:
79
+ raise EvaluationError(
80
+ f"Evaluator failed for dimension '{dimension}': {e}"
81
+ ) from e
82
+
83
+ # Safety check: Ensure the evaluator returned the correct dimension score
84
+ if score.dimension != dimension:
85
+ raise EvaluationError(
86
+ f"Evaluator returned score for dimension '{score.dimension}' "
87
+ f"instead of the expected dimension '{dimension}'."
88
+ )
89
+
90
+ dimension_scores.append(score)
91
+ weighted_score_sum += score.score * weight
92
+ total_weight += weight
93
+
94
+ if total_weight <= 0:
95
+ raise EvaluationError(
96
+ "Total weight of evaluation dimensions must be greater than zero."
97
+ )
98
+
99
+ overall_score = weighted_score_sum / total_weight
100
+ passed = overall_score >= profile.pass_threshold
101
+
102
+ eval_result = EvaluationResult(
103
+ benchmark_id=benchmark.benchmark_id,
104
+ benchmark_name=benchmark.name,
105
+ overall_score=overall_score,
106
+ dimension_scores=dimension_scores,
107
+ passed=passed,
108
+ )
109
+
110
+ return eval_result
@@ -0,0 +1,7 @@
1
+ """Package containing concrete evaluation dimension strategies."""
2
+
3
+ from framework.evaluation.evaluators.constraint import ConstraintEvaluator
4
+ from framework.evaluation.evaluators.planning import PlanningQualityEvaluator
5
+ from framework.evaluation.evaluators.personalization import PersonalizationEvaluator
6
+ from framework.evaluation.evaluators.adaptability import AdaptabilityEvaluator
7
+ from framework.evaluation.evaluators.information_accuracy import InformationAccuracyEvaluator
@@ -0,0 +1,27 @@
1
+ """Evaluator for adaptability quality using LLM judgment."""
2
+
3
+ from typing import List
4
+ from framework.models import AgentOutput, Benchmark
5
+ from framework.evaluation.evaluators.base_llm import BaseLLMEvaluator
6
+ from framework.evaluation.prompts import build_adaptability_prompt
7
+ from framework.evaluation.dimensions import ADAPTABILITY
8
+ from framework.llms import Message
9
+
10
+
11
+ class AdaptabilityEvaluator(BaseLLMEvaluator):
12
+ """Evaluates the 'Adaptability' dimension of an agent's output.
13
+
14
+ Uses an LLM Judge to check how well the itinerary is modified to respond to changing
15
+ circumstances while keeping traveler preferences intact.
16
+ """
17
+
18
+ @property
19
+ def dimension(self) -> str:
20
+ return ADAPTABILITY
21
+
22
+ def build_prompt(
23
+ self,
24
+ benchmark: Benchmark,
25
+ output: AgentOutput,
26
+ ) -> List[Message]:
27
+ return build_adaptability_prompt(benchmark, output)
@@ -0,0 +1,104 @@
1
+ """Base class for all LLM-based evaluators."""
2
+
3
+ from abc import abstractmethod
4
+ import json
5
+ from typing import Dict, List
6
+ from framework.evaluation.base import BaseEvaluator
7
+ from framework.exceptions import EvaluationError
8
+ from framework.models import AgentOutput, Benchmark, DimensionScore
9
+ from framework.llms import BaseLLM, Message
10
+ from ...utils import parse_json_markdown
11
+
12
+
13
+ class BaseLLMEvaluator(BaseEvaluator):
14
+ """Abstract base class implementing orchestration and parsing for LLM judges."""
15
+
16
+ def __init__(self, llm: BaseLLM):
17
+ """Initializes the BaseLLMEvaluator.
18
+
19
+ Args:
20
+ llm: An instance of a BaseLLM provider wrapper.
21
+ """
22
+ self.llm = llm
23
+
24
+ @property
25
+ @abstractmethod
26
+ def dimension(self) -> str:
27
+ """The dimension name this evaluator evaluates.
28
+
29
+ Returns:
30
+ A string containing the evaluation dimension name.
31
+ """
32
+ pass
33
+
34
+ @abstractmethod
35
+ def build_prompt(
36
+ self, benchmark: Benchmark, output: AgentOutput
37
+ ) -> List[Message]:
38
+ """Constructs the prompt messages for the LLM.
39
+
40
+ Args:
41
+ benchmark: The scenario benchmark details.
42
+ output: The response/data generated by the AI agent.
43
+
44
+ Returns:
45
+ A list of Message objects.
46
+ """
47
+ pass
48
+
49
+ def _parse_json_response(self, text: str) -> dict:
50
+ """Extracts and parses a JSON object from raw LLM output.
51
+
52
+ Handles wrapping blocks like ```json ... ``` robustly.
53
+
54
+ Args:
55
+ text: Raw text generated by the model.
56
+
57
+ Returns:
58
+ The parsed dictionary.
59
+
60
+ Raises:
61
+ ValueError: If parsing fails.
62
+ """
63
+ parsed = parse_json_markdown(text)
64
+ if not isinstance(parsed, dict):
65
+ raise ValueError(f"Expected JSON dictionary, got: {type(parsed)}")
66
+ return parsed
67
+
68
+ def evaluate(
69
+ self,
70
+ benchmark: Benchmark,
71
+ output: AgentOutput,
72
+ ) -> DimensionScore:
73
+ """Orchestrates LLM prompt execution, response parsing, and score wrapping.
74
+
75
+ Args:
76
+ benchmark: The scenario benchmark details.
77
+ output: The response/data generated by the AI agent.
78
+
79
+ Returns:
80
+ A DimensionScore containing the final rating.
81
+
82
+ Raises:
83
+ EvaluationError: If the model generation or JSON parsing fails.
84
+ """
85
+ messages = self.build_prompt(benchmark, output)
86
+ response_text = ""
87
+
88
+ try:
89
+ response = self.llm.generate(messages)
90
+ response_text = response.text
91
+ parsed = self._parse_json_response(response_text)
92
+ score = float(parsed["score"])
93
+ reason = str(parsed["reason"])
94
+ except Exception as e:
95
+ raise EvaluationError(
96
+ f"Failed to generate or parse LLM evaluation response: {e}. "
97
+ f"Raw response: {response_text}"
98
+ ) from e
99
+
100
+ return DimensionScore(
101
+ dimension=self.dimension,
102
+ score=score,
103
+ reason=reason,
104
+ )
@@ -0,0 +1,27 @@
1
+ """Evaluator for constraint satisfaction using LLM judgment."""
2
+
3
+ from typing import List
4
+ from framework.models import AgentOutput, Benchmark
5
+ from framework.evaluation.evaluators.base_llm import BaseLLMEvaluator
6
+ from framework.evaluation.prompts import build_constraint_prompt
7
+ from framework.evaluation.dimensions import CONSTRAINT_SATISFACTION
8
+ from framework.llms import Message
9
+
10
+
11
+ class ConstraintEvaluator(BaseLLMEvaluator):
12
+ """Evaluates the 'Constraint Satisfaction' dimension of an agent's output.
13
+
14
+ Uses an LLM Judge to determine if all constraints defined in the benchmark
15
+ scenario are successfully met.
16
+ """
17
+
18
+ @property
19
+ def dimension(self) -> str:
20
+ return CONSTRAINT_SATISFACTION
21
+
22
+ def build_prompt(
23
+ self,
24
+ benchmark: Benchmark,
25
+ output: AgentOutput,
26
+ ) -> List[Message]:
27
+ return build_constraint_prompt(benchmark, output)
@@ -0,0 +1,41 @@
1
+ """Evaluator for factual information accuracy using hybrid verifiers."""
2
+
3
+ from typing import List
4
+ from framework.evaluation.evaluators.base_llm import BaseLLMEvaluator
5
+ from framework.evaluation.prompts.information_accuracy import build_information_accuracy_prompt
6
+ from framework.evaluation.dimensions import INFORMATION_ACCURACY
7
+ from framework.llms import BaseLLM, Message
8
+ from framework.models import AgentOutput, Benchmark
9
+ from framework.verification.pipeline import VerificationPipeline
10
+
11
+
12
+ class InformationAccuracyEvaluator(BaseLLMEvaluator):
13
+ """Evaluates the 'Information Accuracy' dimension of an agent's output.
14
+
15
+ Uses an injected VerificationPipeline to check factual claims,
16
+ and formats them to get graded by the LLM Judge.
17
+ """
18
+
19
+ def __init__(self, llm: BaseLLM, pipeline: VerificationPipeline):
20
+ """Initializes the InformationAccuracyEvaluator.
21
+
22
+ Args:
23
+ llm: The LLM client for final evidence grading.
24
+ pipeline: The VerificationPipeline instance to verify claims.
25
+ """
26
+ super().__init__(llm)
27
+ self.pipeline = pipeline
28
+
29
+ @property
30
+ def dimension(self) -> str:
31
+ return INFORMATION_ACCURACY
32
+
33
+ def build_prompt(
34
+ self,
35
+ benchmark: Benchmark,
36
+ output: AgentOutput,
37
+ ) -> List[Message]:
38
+ # 1. Run the injected pipeline to extract and verify factual claims
39
+ report = self.pipeline.run(output)
40
+ # 2. Format findings context and return prompts list
41
+ return build_information_accuracy_prompt(benchmark, output, report)
@@ -0,0 +1,27 @@
1
+ """Evaluator for personalization quality using LLM judgment."""
2
+
3
+ from typing import List
4
+ from framework.models import AgentOutput, Benchmark
5
+ from framework.evaluation.evaluators.base_llm import BaseLLMEvaluator
6
+ from framework.evaluation.prompts import build_personalization_prompt
7
+ from framework.evaluation.dimensions import PERSONALIZATION
8
+ from framework.llms import Message
9
+
10
+
11
+ class PersonalizationEvaluator(BaseLLMEvaluator):
12
+ """Evaluates the 'Personalization' dimension of an agent's output.
13
+
14
+ Uses an LLM Judge to check if the itinerary is tailored to the traveler's stated
15
+ interests, hobbies, remote working requirements, and pacing preferences.
16
+ """
17
+
18
+ @property
19
+ def dimension(self) -> str:
20
+ return PERSONALIZATION
21
+
22
+ def build_prompt(
23
+ self,
24
+ benchmark: Benchmark,
25
+ output: AgentOutput,
26
+ ) -> List[Message]:
27
+ return build_personalization_prompt(benchmark, output)
@@ -0,0 +1,27 @@
1
+ """Evaluator for planning quality using LLM judgment."""
2
+
3
+ from typing import List
4
+ from framework.models import AgentOutput, Benchmark
5
+ from framework.evaluation.evaluators.base_llm import BaseLLMEvaluator
6
+ from framework.evaluation.prompts import build_planning_prompt
7
+ from framework.evaluation.dimensions import PLANNING_QUALITY
8
+ from framework.llms import Message
9
+
10
+
11
+ class PlanningQualityEvaluator(BaseLLMEvaluator):
12
+ """Evaluates the 'Planning Quality' dimension of an agent's output.
13
+
14
+ Uses an LLM Judge to determine if the itinerary follows a logical,
15
+ efficient route, minimizes accommodation changes, and balances pacing.
16
+ """
17
+
18
+ @property
19
+ def dimension(self) -> str:
20
+ return PLANNING_QUALITY
21
+
22
+ def build_prompt(
23
+ self,
24
+ benchmark: Benchmark,
25
+ output: AgentOutput,
26
+ ) -> List[Message]:
27
+ return build_planning_prompt(benchmark, output)
@@ -0,0 +1,81 @@
1
+ """Support-domain evaluator prompts.
2
+
3
+ The core dimensions are shared with travel, but their rubrics are not. Keeping
4
+ these prompts separate prevents a support ticket from being judged as if it
5
+ were an itinerary.
6
+ """
7
+
8
+ from typing import List
9
+
10
+ from framework.evaluation.dimensions import (
11
+ ADAPTABILITY,
12
+ INFORMATION_ACCURACY,
13
+ PERSONALIZATION,
14
+ PLANNING_QUALITY,
15
+ )
16
+ from framework.evaluation.evaluators.base_llm import BaseLLMEvaluator
17
+ from framework.evaluation.prompts.base import build_llm_judge_prompt
18
+ from framework.llms import Message
19
+ from framework.models import AgentOutput, Benchmark
20
+ from framework.verification.pipeline import VerificationPipeline
21
+
22
+
23
+ class _SupportEvaluator(BaseLLMEvaluator):
24
+ rubric = ""
25
+
26
+ def build_prompt(self, benchmark: Benchmark, output: AgentOutput) -> List[Message]:
27
+ return build_llm_judge_prompt(
28
+ "You are an objective evaluator for a customer-support triage agent.",
29
+ self.rubric,
30
+ benchmark,
31
+ output,
32
+ )
33
+
34
+
35
+ class SupportPlanningEvaluator(_SupportEvaluator):
36
+ dimension = PLANNING_QUALITY
37
+ rubric = (
38
+ "Evaluate whether the triage workflow is ordered and operationally useful. "
39
+ "Check acknowledgement, SLA preservation, escalation order, evidence gathering, "
40
+ "containment actions, and the next customer update. Do not apply travel or itinerary criteria."
41
+ )
42
+
43
+
44
+ class SupportPersonalizationEvaluator(_SupportEvaluator):
45
+ dimension = PERSONALIZATION
46
+ rubric = (
47
+ "Evaluate whether the response is appropriately tailored to this enterprise customer, "
48
+ "the EU production impact, the imminent launch, and the requested support-owner role. "
49
+ "Assess empathy, clarity, and usefulness of the customer-facing message."
50
+ )
51
+
52
+
53
+ class SupportAdaptabilityEvaluator(_SupportEvaluator):
54
+ dimension = ADAPTABILITY
55
+ rubric = (
56
+ "Evaluate how well the triage plan handles uncertainty and branching conditions. "
57
+ "Check the distinction between immediate Payments/Incident Commander escalation and "
58
+ "conditional EU platform escalation, while avoiding an unverified root-cause claim."
59
+ )
60
+
61
+
62
+ class SupportInformationAccuracyEvaluator(BaseLLMEvaluator):
63
+ dimension = INFORMATION_ACCURACY
64
+
65
+ def __init__(self, llm, pipeline: VerificationPipeline):
66
+ super().__init__(llm)
67
+ self.pipeline = pipeline
68
+
69
+ def build_prompt(self, benchmark: Benchmark, output: AgentOutput) -> List[Message]:
70
+ report = self.pipeline.run(output)
71
+ return build_llm_judge_prompt(
72
+ "You are an objective evaluator checking factual accuracy in a customer-support triage response.",
73
+ (
74
+ "Check that observed incident facts, priority, timestamps, impact, and escalation "
75
+ "targets match the scenario. Separate facts from hypotheses and do not penalize "
76
+ "unknown claims merely because they are absent from an unrelated knowledge base.\n\n"
77
+ f"Verification evidence:\n{report}"
78
+ ),
79
+ benchmark,
80
+ output,
81
+ )
@@ -0,0 +1,11 @@
1
+ """Evaluation prompts package."""
2
+
3
+ from framework.evaluation.prompts.base import (
4
+ JSON_RESPONSE_SCHEMA,
5
+ build_llm_judge_prompt,
6
+ )
7
+ from framework.evaluation.prompts.constraint import build_constraint_prompt
8
+ from framework.evaluation.prompts.planning import build_planning_prompt
9
+ from framework.evaluation.prompts.personalization import build_personalization_prompt
10
+ from framework.evaluation.prompts.adaptability import build_adaptability_prompt
11
+ from framework.evaluation.prompts.information_accuracy import build_information_accuracy_prompt