evalrun 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agents/__init__.py +6 -0
- agents/auditor/__init__.py +13 -0
- agents/auditor/budget_auditor.py +91 -0
- agents/auditor/parser.py +139 -0
- agents/auditor/prompts.py +64 -0
- agents/auditor/schema.py +67 -0
- agents/base.py +20 -0
- agents/reflection/__init__.py +3 -0
- agents/reflection/agent.py +77 -0
- agents/reflection/prompts.py +18 -0
- agents/research/__init__.py +4 -0
- agents/research/agent.py +45 -0
- agents/research/planner.py +52 -0
- agents/research/prompts.py +14 -0
- agents/support/__init__.py +5 -0
- agents/support/triage_agent.py +45 -0
- agents/travel/__init__.py +11 -0
- agents/travel/agent.py +377 -0
- agents/travel/prompts.py +30 -0
- agents/travel/session.py +110 -0
- cli/__init__.py +6 -0
- cli/demo.py +47 -0
- cli/formatter.py +93 -0
- cli/html_reporter.py +647 -0
- cli/main.py +423 -0
- cli/progress.py +38 -0
- cli/resolver.py +99 -0
- evalrun-0.4.0.dist-info/METADATA +268 -0
- evalrun-0.4.0.dist-info/RECORD +100 -0
- evalrun-0.4.0.dist-info/WHEEL +5 -0
- evalrun-0.4.0.dist-info/entry_points.txt +2 -0
- evalrun-0.4.0.dist-info/licenses/LICENSE +0 -0
- evalrun-0.4.0.dist-info/top_level.txt +4 -0
- framework/__init__.py +70 -0
- framework/core/__init__.py +17 -0
- framework/core/adapters.py +118 -0
- framework/core/contracts.py +88 -0
- framework/core/suite.py +44 -0
- framework/evaluation/__init__.py +22 -0
- framework/evaluation/base.py +29 -0
- framework/evaluation/dimensions.py +7 -0
- framework/evaluation/engine.py +110 -0
- framework/evaluation/evaluators/__init__.py +7 -0
- framework/evaluation/evaluators/adaptability.py +27 -0
- framework/evaluation/evaluators/base_llm.py +104 -0
- framework/evaluation/evaluators/constraint.py +27 -0
- framework/evaluation/evaluators/information_accuracy.py +41 -0
- framework/evaluation/evaluators/personalization.py +27 -0
- framework/evaluation/evaluators/planning.py +27 -0
- framework/evaluation/evaluators/support.py +81 -0
- framework/evaluation/prompts/__init__.py +11 -0
- framework/evaluation/prompts/adaptability.py +57 -0
- framework/evaluation/prompts/base.py +52 -0
- framework/evaluation/prompts/constraint.py +41 -0
- framework/evaluation/prompts/information_accuracy.py +79 -0
- framework/evaluation/prompts/personalization.py +57 -0
- framework/evaluation/prompts/planning.py +61 -0
- framework/evaluation/runner.py +363 -0
- framework/evaluation/testing.py +25 -0
- framework/exceptions.py +49 -0
- framework/llms/__init__.py +8 -0
- framework/llms/base.py +37 -0
- framework/llms/factory.py +38 -0
- framework/llms/gemini.py +85 -0
- framework/llms/mock.py +25 -0
- framework/llms/openai.py +94 -0
- framework/llms/openai_compatible.py +139 -0
- framework/mcp/__init__.py +20 -0
- framework/mcp/client.py +62 -0
- framework/mcp/constraints.py +125 -0
- framework/mcp/revision_summary.py +122 -0
- framework/mcp/server.py +49 -0
- framework/memory/__init__.py +3 -0
- framework/memory/base.py +17 -0
- framework/models.py +83 -0
- framework/parser.py +45 -0
- framework/parsers/__init__.py +12 -0
- framework/parsers/frontmatter.py +28 -0
- framework/parsers/mapper.py +66 -0
- framework/parsers/markdown.py +122 -0
- framework/parsers/transformers.py +112 -0
- framework/profiles/__init__.py +28 -0
- framework/profiles/registry.py +104 -0
- framework/profiles/support.py +27 -0
- framework/profiles/travel.py +78 -0
- framework/regression/__init__.py +17 -0
- framework/regression/comparator.py +273 -0
- framework/regression/loader.py +145 -0
- framework/sdk.py +151 -0
- framework/utils.py +41 -0
- framework/verification/__init__.py +13 -0
- framework/verification/base.py +32 -0
- framework/verification/extractor.py +105 -0
- framework/verification/local.py +122 -0
- framework/verification/models.py +112 -0
- framework/verification/pipeline.py +36 -0
- framework/verification/prompts.py +24 -0
- framework/verification/utils.py +54 -0
- ui/__init__.py +1 -0
- ui/server.py +255 -0
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""Domain-neutral core contracts and data models for evaluation."""
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
from datetime import datetime, timezone
|
|
5
|
+
from typing import Any, Dict, List, Literal, Optional
|
|
6
|
+
from framework.models import Benchmark
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass
|
|
10
|
+
class Scenario:
|
|
11
|
+
"""Domain-neutral evaluation scenario specification."""
|
|
12
|
+
|
|
13
|
+
id: str
|
|
14
|
+
name: str
|
|
15
|
+
domain: str # e.g. "travel", "support_triage", "scheduling"
|
|
16
|
+
description: str
|
|
17
|
+
prompt: str
|
|
18
|
+
constraints: Dict[str, Any]
|
|
19
|
+
expected_behavior: List[str]
|
|
20
|
+
evaluation_criteria: Dict[str, List[str]]
|
|
21
|
+
pass_criteria: List[str]
|
|
22
|
+
failure_conditions: List[str]
|
|
23
|
+
notes: Optional[List[str]] = None
|
|
24
|
+
profile_name: str = "travel-agent"
|
|
25
|
+
metadata: Dict[str, Any] = field(default_factory=dict)
|
|
26
|
+
|
|
27
|
+
@property
|
|
28
|
+
def benchmark_id(self) -> str:
|
|
29
|
+
"""Backward-compatibility property for legacy codebase."""
|
|
30
|
+
return self.id
|
|
31
|
+
|
|
32
|
+
@property
|
|
33
|
+
def profile(self) -> str:
|
|
34
|
+
"""Backward-compatibility property for legacy profile name."""
|
|
35
|
+
return self.profile_name
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def to_scenario(benchmark: Benchmark, domain: str = "travel") -> Scenario:
|
|
39
|
+
"""Converts a legacy Benchmark instance into a generic Scenario instance."""
|
|
40
|
+
return Scenario(
|
|
41
|
+
id=benchmark.benchmark_id,
|
|
42
|
+
name=benchmark.name,
|
|
43
|
+
domain=domain,
|
|
44
|
+
description=benchmark.description,
|
|
45
|
+
prompt=benchmark.prompt,
|
|
46
|
+
constraints=benchmark.constraints,
|
|
47
|
+
expected_behavior=benchmark.expected_behavior,
|
|
48
|
+
evaluation_criteria=benchmark.evaluation_criteria,
|
|
49
|
+
pass_criteria=benchmark.pass_criteria,
|
|
50
|
+
failure_conditions=benchmark.failure_conditions,
|
|
51
|
+
notes=benchmark.notes,
|
|
52
|
+
profile_name=getattr(benchmark, "profile", "travel-agent"),
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def to_benchmark(scenario: Scenario) -> Benchmark:
|
|
57
|
+
"""Converts a generic Scenario instance into a legacy Benchmark instance."""
|
|
58
|
+
return Benchmark(
|
|
59
|
+
benchmark_id=scenario.id,
|
|
60
|
+
name=scenario.name,
|
|
61
|
+
description=scenario.description,
|
|
62
|
+
prompt=scenario.prompt,
|
|
63
|
+
constraints=scenario.constraints,
|
|
64
|
+
expected_behavior=scenario.expected_behavior,
|
|
65
|
+
evaluation_criteria=scenario.evaluation_criteria,
|
|
66
|
+
pass_criteria=scenario.pass_criteria,
|
|
67
|
+
failure_conditions=scenario.failure_conditions,
|
|
68
|
+
notes=scenario.notes or [],
|
|
69
|
+
profile=scenario.profile_name,
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
@dataclass
|
|
74
|
+
class RunTrace:
|
|
75
|
+
"""Execution trace tracking metadata, latency, token usage, and status for an evaluation run."""
|
|
76
|
+
|
|
77
|
+
trace_id: str
|
|
78
|
+
scenario_id: str
|
|
79
|
+
agent_id: str
|
|
80
|
+
model_name: str
|
|
81
|
+
started_at_utc: str
|
|
82
|
+
finished_at_utc: str
|
|
83
|
+
latency_seconds: float
|
|
84
|
+
status: Literal["success", "error", "timeout"]
|
|
85
|
+
error: Optional[str] = None
|
|
86
|
+
retries_attempted: int = 0
|
|
87
|
+
token_usage: Optional[Dict[str, int]] = None
|
|
88
|
+
metadata: Dict[str, Any] = field(default_factory=dict)
|
framework/core/suite.py
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""EvaluationSuite class managing versioned collections of Scenarios and EvaluationProfiles."""
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
from typing import Dict, List, Optional
|
|
5
|
+
|
|
6
|
+
from framework.core.contracts import Scenario
|
|
7
|
+
from framework.models import EvaluationProfile
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass
|
|
11
|
+
class EvaluationSuite:
|
|
12
|
+
"""Encapsulates a collection of scenarios and domain evaluation profiles."""
|
|
13
|
+
|
|
14
|
+
suite_id: str
|
|
15
|
+
name: str
|
|
16
|
+
domain: str
|
|
17
|
+
version: str
|
|
18
|
+
scenarios: List[Scenario]
|
|
19
|
+
profiles: Dict[str, EvaluationProfile] = field(default_factory=dict)
|
|
20
|
+
|
|
21
|
+
def get_scenario(self, scenario_id: str) -> Optional[Scenario]:
|
|
22
|
+
"""Looks up a scenario by ID."""
|
|
23
|
+
for scenario in self.scenarios:
|
|
24
|
+
if scenario.id == scenario_id:
|
|
25
|
+
return scenario
|
|
26
|
+
return None
|
|
27
|
+
|
|
28
|
+
def get_profile(self, profile_name: Optional[str] = None) -> EvaluationProfile:
|
|
29
|
+
"""Resolves an evaluation profile by name or default key."""
|
|
30
|
+
if profile_name:
|
|
31
|
+
if profile_name in self.profiles:
|
|
32
|
+
return self.profiles[profile_name]
|
|
33
|
+
available = list(self.profiles.keys())
|
|
34
|
+
raise ValueError(
|
|
35
|
+
f"Profile '{profile_name}' not found in suite '{self.suite_id}'. "
|
|
36
|
+
f"Available profiles: {available}"
|
|
37
|
+
)
|
|
38
|
+
if "default" in self.profiles:
|
|
39
|
+
return self.profiles["default"]
|
|
40
|
+
available = list(self.profiles.keys())
|
|
41
|
+
raise ValueError(
|
|
42
|
+
f"No profile specified and no 'default' profile registered for suite '{self.suite_id}'. "
|
|
43
|
+
f"Available profiles: {available}"
|
|
44
|
+
)
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""Package containing the evaluation engine, interfaces, and concrete evaluators."""
|
|
2
|
+
|
|
3
|
+
from framework.evaluation.base import BaseEvaluator
|
|
4
|
+
from framework.evaluation.engine import EvaluationEngine
|
|
5
|
+
from framework.evaluation.runner import BenchmarkRunner
|
|
6
|
+
from framework.llms import BaseLLM, MockLLM, OpenAILLM, GeminiLLM
|
|
7
|
+
from framework.evaluation.evaluators import (
|
|
8
|
+
ConstraintEvaluator,
|
|
9
|
+
PlanningQualityEvaluator,
|
|
10
|
+
PersonalizationEvaluator,
|
|
11
|
+
AdaptabilityEvaluator,
|
|
12
|
+
InformationAccuracyEvaluator,
|
|
13
|
+
)
|
|
14
|
+
from framework.evaluation.evaluators.base_llm import BaseLLMEvaluator
|
|
15
|
+
from framework.evaluation.testing import DummyEvaluator
|
|
16
|
+
from framework.evaluation.dimensions import (
|
|
17
|
+
CONSTRAINT_SATISFACTION,
|
|
18
|
+
PLANNING_QUALITY,
|
|
19
|
+
INFORMATION_ACCURACY,
|
|
20
|
+
PERSONALIZATION,
|
|
21
|
+
ADAPTABILITY,
|
|
22
|
+
)
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""Base interface for all evaluation strategies."""
|
|
2
|
+
|
|
3
|
+
from abc import ABC, abstractmethod
|
|
4
|
+
from framework.models import AgentOutput, Benchmark, DimensionScore
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class BaseEvaluator(ABC):
|
|
8
|
+
"""Abstract base class defining the interface for evaluation strategies.
|
|
9
|
+
|
|
10
|
+
Evaluators assess an agent's output against a benchmark scenario for a
|
|
11
|
+
single specific evaluation dimension.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
@abstractmethod
|
|
15
|
+
def evaluate(
|
|
16
|
+
self,
|
|
17
|
+
benchmark: Benchmark,
|
|
18
|
+
output: AgentOutput,
|
|
19
|
+
) -> DimensionScore:
|
|
20
|
+
"""Evaluates a single dimension of an agent's output against a benchmark.
|
|
21
|
+
|
|
22
|
+
Args:
|
|
23
|
+
benchmark: The scenario benchmark details.
|
|
24
|
+
output: The response/data generated by the AI agent.
|
|
25
|
+
|
|
26
|
+
Returns:
|
|
27
|
+
A DimensionScore containing the dimension name, score, and reasoning.
|
|
28
|
+
"""
|
|
29
|
+
pass
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
from langfuse import observe
|
|
2
|
+
|
|
3
|
+
from typing import Dict
|
|
4
|
+
from framework.exceptions import EvaluationError
|
|
5
|
+
from framework.models import (
|
|
6
|
+
AgentOutput,
|
|
7
|
+
Benchmark,
|
|
8
|
+
DimensionScore,
|
|
9
|
+
EvaluationProfile,
|
|
10
|
+
EvaluationResult,
|
|
11
|
+
)
|
|
12
|
+
from framework.evaluation.base import BaseEvaluator
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class EvaluationEngine:
|
|
16
|
+
"""Orchestrates the evaluation of an agent's output against a benchmark scenario.
|
|
17
|
+
|
|
18
|
+
Uses an EvaluationProfile to weigh scores across multiple dimensions and
|
|
19
|
+
calculates the final evaluation outcome.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
def __init__(self, evaluators: Dict[str, BaseEvaluator]):
|
|
23
|
+
"""Initializes the EvaluationEngine.
|
|
24
|
+
|
|
25
|
+
Args:
|
|
26
|
+
evaluators: A dictionary mapping dimension names to BaseEvaluator instances.
|
|
27
|
+
"""
|
|
28
|
+
self.evaluators = evaluators
|
|
29
|
+
|
|
30
|
+
@observe(name="evaluation-engine")
|
|
31
|
+
def evaluate(
|
|
32
|
+
self,
|
|
33
|
+
benchmark: Benchmark,
|
|
34
|
+
output: AgentOutput,
|
|
35
|
+
profile: EvaluationProfile,
|
|
36
|
+
) -> EvaluationResult:
|
|
37
|
+
"""Runs the evaluation pipeline for the given benchmark and agent output.
|
|
38
|
+
|
|
39
|
+
Args:
|
|
40
|
+
benchmark: The benchmark scenario.
|
|
41
|
+
output: The response/data generated by the AI agent.
|
|
42
|
+
profile: The evaluation profile specifying dimension weights.
|
|
43
|
+
|
|
44
|
+
Returns:
|
|
45
|
+
An EvaluationResult containing the dimension breakdown and final status.
|
|
46
|
+
|
|
47
|
+
Raises:
|
|
48
|
+
EvaluationError: If a dimension specified in the profile has no registered evaluator,
|
|
49
|
+
is missing from the benchmark criteria, or returns an inconsistent dimension name.
|
|
50
|
+
"""
|
|
51
|
+
if not profile.weights:
|
|
52
|
+
raise EvaluationError(
|
|
53
|
+
f"Evaluation profile '{profile.name}' contains no weights."
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
# Validate that all profile dimensions exist in the benchmark criteria
|
|
57
|
+
for dimension in profile.weights.keys():
|
|
58
|
+
if dimension not in benchmark.evaluation_criteria:
|
|
59
|
+
raise EvaluationError(
|
|
60
|
+
f"Dimension '{dimension}' from profile '{profile.name}' "
|
|
61
|
+
f"is not defined in benchmark '{benchmark.benchmark_id}' evaluation criteria."
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
dimension_scores = []
|
|
65
|
+
weighted_score_sum = 0.0
|
|
66
|
+
total_weight = 0.0
|
|
67
|
+
|
|
68
|
+
for dimension, weight in profile.weights.items():
|
|
69
|
+
evaluator = self.evaluators.get(dimension)
|
|
70
|
+
if not evaluator:
|
|
71
|
+
raise EvaluationError(
|
|
72
|
+
f"No evaluator registered for dimension '{dimension}' "
|
|
73
|
+
f"in profile '{profile.name}'."
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
try:
|
|
77
|
+
score = evaluator.evaluate(benchmark, output)
|
|
78
|
+
except Exception as e:
|
|
79
|
+
raise EvaluationError(
|
|
80
|
+
f"Evaluator failed for dimension '{dimension}': {e}"
|
|
81
|
+
) from e
|
|
82
|
+
|
|
83
|
+
# Safety check: Ensure the evaluator returned the correct dimension score
|
|
84
|
+
if score.dimension != dimension:
|
|
85
|
+
raise EvaluationError(
|
|
86
|
+
f"Evaluator returned score for dimension '{score.dimension}' "
|
|
87
|
+
f"instead of the expected dimension '{dimension}'."
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
dimension_scores.append(score)
|
|
91
|
+
weighted_score_sum += score.score * weight
|
|
92
|
+
total_weight += weight
|
|
93
|
+
|
|
94
|
+
if total_weight <= 0:
|
|
95
|
+
raise EvaluationError(
|
|
96
|
+
"Total weight of evaluation dimensions must be greater than zero."
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
overall_score = weighted_score_sum / total_weight
|
|
100
|
+
passed = overall_score >= profile.pass_threshold
|
|
101
|
+
|
|
102
|
+
eval_result = EvaluationResult(
|
|
103
|
+
benchmark_id=benchmark.benchmark_id,
|
|
104
|
+
benchmark_name=benchmark.name,
|
|
105
|
+
overall_score=overall_score,
|
|
106
|
+
dimension_scores=dimension_scores,
|
|
107
|
+
passed=passed,
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
return eval_result
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
"""Package containing concrete evaluation dimension strategies."""
|
|
2
|
+
|
|
3
|
+
from framework.evaluation.evaluators.constraint import ConstraintEvaluator
|
|
4
|
+
from framework.evaluation.evaluators.planning import PlanningQualityEvaluator
|
|
5
|
+
from framework.evaluation.evaluators.personalization import PersonalizationEvaluator
|
|
6
|
+
from framework.evaluation.evaluators.adaptability import AdaptabilityEvaluator
|
|
7
|
+
from framework.evaluation.evaluators.information_accuracy import InformationAccuracyEvaluator
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Evaluator for adaptability quality using LLM judgment."""
|
|
2
|
+
|
|
3
|
+
from typing import List
|
|
4
|
+
from framework.models import AgentOutput, Benchmark
|
|
5
|
+
from framework.evaluation.evaluators.base_llm import BaseLLMEvaluator
|
|
6
|
+
from framework.evaluation.prompts import build_adaptability_prompt
|
|
7
|
+
from framework.evaluation.dimensions import ADAPTABILITY
|
|
8
|
+
from framework.llms import Message
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class AdaptabilityEvaluator(BaseLLMEvaluator):
|
|
12
|
+
"""Evaluates the 'Adaptability' dimension of an agent's output.
|
|
13
|
+
|
|
14
|
+
Uses an LLM Judge to check how well the itinerary is modified to respond to changing
|
|
15
|
+
circumstances while keeping traveler preferences intact.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
@property
|
|
19
|
+
def dimension(self) -> str:
|
|
20
|
+
return ADAPTABILITY
|
|
21
|
+
|
|
22
|
+
def build_prompt(
|
|
23
|
+
self,
|
|
24
|
+
benchmark: Benchmark,
|
|
25
|
+
output: AgentOutput,
|
|
26
|
+
) -> List[Message]:
|
|
27
|
+
return build_adaptability_prompt(benchmark, output)
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""Base class for all LLM-based evaluators."""
|
|
2
|
+
|
|
3
|
+
from abc import abstractmethod
|
|
4
|
+
import json
|
|
5
|
+
from typing import Dict, List
|
|
6
|
+
from framework.evaluation.base import BaseEvaluator
|
|
7
|
+
from framework.exceptions import EvaluationError
|
|
8
|
+
from framework.models import AgentOutput, Benchmark, DimensionScore
|
|
9
|
+
from framework.llms import BaseLLM, Message
|
|
10
|
+
from ...utils import parse_json_markdown
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class BaseLLMEvaluator(BaseEvaluator):
|
|
14
|
+
"""Abstract base class implementing orchestration and parsing for LLM judges."""
|
|
15
|
+
|
|
16
|
+
def __init__(self, llm: BaseLLM):
|
|
17
|
+
"""Initializes the BaseLLMEvaluator.
|
|
18
|
+
|
|
19
|
+
Args:
|
|
20
|
+
llm: An instance of a BaseLLM provider wrapper.
|
|
21
|
+
"""
|
|
22
|
+
self.llm = llm
|
|
23
|
+
|
|
24
|
+
@property
|
|
25
|
+
@abstractmethod
|
|
26
|
+
def dimension(self) -> str:
|
|
27
|
+
"""The dimension name this evaluator evaluates.
|
|
28
|
+
|
|
29
|
+
Returns:
|
|
30
|
+
A string containing the evaluation dimension name.
|
|
31
|
+
"""
|
|
32
|
+
pass
|
|
33
|
+
|
|
34
|
+
@abstractmethod
|
|
35
|
+
def build_prompt(
|
|
36
|
+
self, benchmark: Benchmark, output: AgentOutput
|
|
37
|
+
) -> List[Message]:
|
|
38
|
+
"""Constructs the prompt messages for the LLM.
|
|
39
|
+
|
|
40
|
+
Args:
|
|
41
|
+
benchmark: The scenario benchmark details.
|
|
42
|
+
output: The response/data generated by the AI agent.
|
|
43
|
+
|
|
44
|
+
Returns:
|
|
45
|
+
A list of Message objects.
|
|
46
|
+
"""
|
|
47
|
+
pass
|
|
48
|
+
|
|
49
|
+
def _parse_json_response(self, text: str) -> dict:
|
|
50
|
+
"""Extracts and parses a JSON object from raw LLM output.
|
|
51
|
+
|
|
52
|
+
Handles wrapping blocks like ```json ... ``` robustly.
|
|
53
|
+
|
|
54
|
+
Args:
|
|
55
|
+
text: Raw text generated by the model.
|
|
56
|
+
|
|
57
|
+
Returns:
|
|
58
|
+
The parsed dictionary.
|
|
59
|
+
|
|
60
|
+
Raises:
|
|
61
|
+
ValueError: If parsing fails.
|
|
62
|
+
"""
|
|
63
|
+
parsed = parse_json_markdown(text)
|
|
64
|
+
if not isinstance(parsed, dict):
|
|
65
|
+
raise ValueError(f"Expected JSON dictionary, got: {type(parsed)}")
|
|
66
|
+
return parsed
|
|
67
|
+
|
|
68
|
+
def evaluate(
|
|
69
|
+
self,
|
|
70
|
+
benchmark: Benchmark,
|
|
71
|
+
output: AgentOutput,
|
|
72
|
+
) -> DimensionScore:
|
|
73
|
+
"""Orchestrates LLM prompt execution, response parsing, and score wrapping.
|
|
74
|
+
|
|
75
|
+
Args:
|
|
76
|
+
benchmark: The scenario benchmark details.
|
|
77
|
+
output: The response/data generated by the AI agent.
|
|
78
|
+
|
|
79
|
+
Returns:
|
|
80
|
+
A DimensionScore containing the final rating.
|
|
81
|
+
|
|
82
|
+
Raises:
|
|
83
|
+
EvaluationError: If the model generation or JSON parsing fails.
|
|
84
|
+
"""
|
|
85
|
+
messages = self.build_prompt(benchmark, output)
|
|
86
|
+
response_text = ""
|
|
87
|
+
|
|
88
|
+
try:
|
|
89
|
+
response = self.llm.generate(messages)
|
|
90
|
+
response_text = response.text
|
|
91
|
+
parsed = self._parse_json_response(response_text)
|
|
92
|
+
score = float(parsed["score"])
|
|
93
|
+
reason = str(parsed["reason"])
|
|
94
|
+
except Exception as e:
|
|
95
|
+
raise EvaluationError(
|
|
96
|
+
f"Failed to generate or parse LLM evaluation response: {e}. "
|
|
97
|
+
f"Raw response: {response_text}"
|
|
98
|
+
) from e
|
|
99
|
+
|
|
100
|
+
return DimensionScore(
|
|
101
|
+
dimension=self.dimension,
|
|
102
|
+
score=score,
|
|
103
|
+
reason=reason,
|
|
104
|
+
)
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Evaluator for constraint satisfaction using LLM judgment."""
|
|
2
|
+
|
|
3
|
+
from typing import List
|
|
4
|
+
from framework.models import AgentOutput, Benchmark
|
|
5
|
+
from framework.evaluation.evaluators.base_llm import BaseLLMEvaluator
|
|
6
|
+
from framework.evaluation.prompts import build_constraint_prompt
|
|
7
|
+
from framework.evaluation.dimensions import CONSTRAINT_SATISFACTION
|
|
8
|
+
from framework.llms import Message
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class ConstraintEvaluator(BaseLLMEvaluator):
|
|
12
|
+
"""Evaluates the 'Constraint Satisfaction' dimension of an agent's output.
|
|
13
|
+
|
|
14
|
+
Uses an LLM Judge to determine if all constraints defined in the benchmark
|
|
15
|
+
scenario are successfully met.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
@property
|
|
19
|
+
def dimension(self) -> str:
|
|
20
|
+
return CONSTRAINT_SATISFACTION
|
|
21
|
+
|
|
22
|
+
def build_prompt(
|
|
23
|
+
self,
|
|
24
|
+
benchmark: Benchmark,
|
|
25
|
+
output: AgentOutput,
|
|
26
|
+
) -> List[Message]:
|
|
27
|
+
return build_constraint_prompt(benchmark, output)
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Evaluator for factual information accuracy using hybrid verifiers."""
|
|
2
|
+
|
|
3
|
+
from typing import List
|
|
4
|
+
from framework.evaluation.evaluators.base_llm import BaseLLMEvaluator
|
|
5
|
+
from framework.evaluation.prompts.information_accuracy import build_information_accuracy_prompt
|
|
6
|
+
from framework.evaluation.dimensions import INFORMATION_ACCURACY
|
|
7
|
+
from framework.llms import BaseLLM, Message
|
|
8
|
+
from framework.models import AgentOutput, Benchmark
|
|
9
|
+
from framework.verification.pipeline import VerificationPipeline
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class InformationAccuracyEvaluator(BaseLLMEvaluator):
|
|
13
|
+
"""Evaluates the 'Information Accuracy' dimension of an agent's output.
|
|
14
|
+
|
|
15
|
+
Uses an injected VerificationPipeline to check factual claims,
|
|
16
|
+
and formats them to get graded by the LLM Judge.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
def __init__(self, llm: BaseLLM, pipeline: VerificationPipeline):
|
|
20
|
+
"""Initializes the InformationAccuracyEvaluator.
|
|
21
|
+
|
|
22
|
+
Args:
|
|
23
|
+
llm: The LLM client for final evidence grading.
|
|
24
|
+
pipeline: The VerificationPipeline instance to verify claims.
|
|
25
|
+
"""
|
|
26
|
+
super().__init__(llm)
|
|
27
|
+
self.pipeline = pipeline
|
|
28
|
+
|
|
29
|
+
@property
|
|
30
|
+
def dimension(self) -> str:
|
|
31
|
+
return INFORMATION_ACCURACY
|
|
32
|
+
|
|
33
|
+
def build_prompt(
|
|
34
|
+
self,
|
|
35
|
+
benchmark: Benchmark,
|
|
36
|
+
output: AgentOutput,
|
|
37
|
+
) -> List[Message]:
|
|
38
|
+
# 1. Run the injected pipeline to extract and verify factual claims
|
|
39
|
+
report = self.pipeline.run(output)
|
|
40
|
+
# 2. Format findings context and return prompts list
|
|
41
|
+
return build_information_accuracy_prompt(benchmark, output, report)
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Evaluator for personalization quality using LLM judgment."""
|
|
2
|
+
|
|
3
|
+
from typing import List
|
|
4
|
+
from framework.models import AgentOutput, Benchmark
|
|
5
|
+
from framework.evaluation.evaluators.base_llm import BaseLLMEvaluator
|
|
6
|
+
from framework.evaluation.prompts import build_personalization_prompt
|
|
7
|
+
from framework.evaluation.dimensions import PERSONALIZATION
|
|
8
|
+
from framework.llms import Message
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class PersonalizationEvaluator(BaseLLMEvaluator):
|
|
12
|
+
"""Evaluates the 'Personalization' dimension of an agent's output.
|
|
13
|
+
|
|
14
|
+
Uses an LLM Judge to check if the itinerary is tailored to the traveler's stated
|
|
15
|
+
interests, hobbies, remote working requirements, and pacing preferences.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
@property
|
|
19
|
+
def dimension(self) -> str:
|
|
20
|
+
return PERSONALIZATION
|
|
21
|
+
|
|
22
|
+
def build_prompt(
|
|
23
|
+
self,
|
|
24
|
+
benchmark: Benchmark,
|
|
25
|
+
output: AgentOutput,
|
|
26
|
+
) -> List[Message]:
|
|
27
|
+
return build_personalization_prompt(benchmark, output)
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Evaluator for planning quality using LLM judgment."""
|
|
2
|
+
|
|
3
|
+
from typing import List
|
|
4
|
+
from framework.models import AgentOutput, Benchmark
|
|
5
|
+
from framework.evaluation.evaluators.base_llm import BaseLLMEvaluator
|
|
6
|
+
from framework.evaluation.prompts import build_planning_prompt
|
|
7
|
+
from framework.evaluation.dimensions import PLANNING_QUALITY
|
|
8
|
+
from framework.llms import Message
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class PlanningQualityEvaluator(BaseLLMEvaluator):
|
|
12
|
+
"""Evaluates the 'Planning Quality' dimension of an agent's output.
|
|
13
|
+
|
|
14
|
+
Uses an LLM Judge to determine if the itinerary follows a logical,
|
|
15
|
+
efficient route, minimizes accommodation changes, and balances pacing.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
@property
|
|
19
|
+
def dimension(self) -> str:
|
|
20
|
+
return PLANNING_QUALITY
|
|
21
|
+
|
|
22
|
+
def build_prompt(
|
|
23
|
+
self,
|
|
24
|
+
benchmark: Benchmark,
|
|
25
|
+
output: AgentOutput,
|
|
26
|
+
) -> List[Message]:
|
|
27
|
+
return build_planning_prompt(benchmark, output)
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""Support-domain evaluator prompts.
|
|
2
|
+
|
|
3
|
+
The core dimensions are shared with travel, but their rubrics are not. Keeping
|
|
4
|
+
these prompts separate prevents a support ticket from being judged as if it
|
|
5
|
+
were an itinerary.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from typing import List
|
|
9
|
+
|
|
10
|
+
from framework.evaluation.dimensions import (
|
|
11
|
+
ADAPTABILITY,
|
|
12
|
+
INFORMATION_ACCURACY,
|
|
13
|
+
PERSONALIZATION,
|
|
14
|
+
PLANNING_QUALITY,
|
|
15
|
+
)
|
|
16
|
+
from framework.evaluation.evaluators.base_llm import BaseLLMEvaluator
|
|
17
|
+
from framework.evaluation.prompts.base import build_llm_judge_prompt
|
|
18
|
+
from framework.llms import Message
|
|
19
|
+
from framework.models import AgentOutput, Benchmark
|
|
20
|
+
from framework.verification.pipeline import VerificationPipeline
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class _SupportEvaluator(BaseLLMEvaluator):
|
|
24
|
+
rubric = ""
|
|
25
|
+
|
|
26
|
+
def build_prompt(self, benchmark: Benchmark, output: AgentOutput) -> List[Message]:
|
|
27
|
+
return build_llm_judge_prompt(
|
|
28
|
+
"You are an objective evaluator for a customer-support triage agent.",
|
|
29
|
+
self.rubric,
|
|
30
|
+
benchmark,
|
|
31
|
+
output,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class SupportPlanningEvaluator(_SupportEvaluator):
|
|
36
|
+
dimension = PLANNING_QUALITY
|
|
37
|
+
rubric = (
|
|
38
|
+
"Evaluate whether the triage workflow is ordered and operationally useful. "
|
|
39
|
+
"Check acknowledgement, SLA preservation, escalation order, evidence gathering, "
|
|
40
|
+
"containment actions, and the next customer update. Do not apply travel or itinerary criteria."
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class SupportPersonalizationEvaluator(_SupportEvaluator):
|
|
45
|
+
dimension = PERSONALIZATION
|
|
46
|
+
rubric = (
|
|
47
|
+
"Evaluate whether the response is appropriately tailored to this enterprise customer, "
|
|
48
|
+
"the EU production impact, the imminent launch, and the requested support-owner role. "
|
|
49
|
+
"Assess empathy, clarity, and usefulness of the customer-facing message."
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class SupportAdaptabilityEvaluator(_SupportEvaluator):
|
|
54
|
+
dimension = ADAPTABILITY
|
|
55
|
+
rubric = (
|
|
56
|
+
"Evaluate how well the triage plan handles uncertainty and branching conditions. "
|
|
57
|
+
"Check the distinction between immediate Payments/Incident Commander escalation and "
|
|
58
|
+
"conditional EU platform escalation, while avoiding an unverified root-cause claim."
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class SupportInformationAccuracyEvaluator(BaseLLMEvaluator):
|
|
63
|
+
dimension = INFORMATION_ACCURACY
|
|
64
|
+
|
|
65
|
+
def __init__(self, llm, pipeline: VerificationPipeline):
|
|
66
|
+
super().__init__(llm)
|
|
67
|
+
self.pipeline = pipeline
|
|
68
|
+
|
|
69
|
+
def build_prompt(self, benchmark: Benchmark, output: AgentOutput) -> List[Message]:
|
|
70
|
+
report = self.pipeline.run(output)
|
|
71
|
+
return build_llm_judge_prompt(
|
|
72
|
+
"You are an objective evaluator checking factual accuracy in a customer-support triage response.",
|
|
73
|
+
(
|
|
74
|
+
"Check that observed incident facts, priority, timestamps, impact, and escalation "
|
|
75
|
+
"targets match the scenario. Separate facts from hypotheses and do not penalize "
|
|
76
|
+
"unknown claims merely because they are absent from an unrelated knowledge base.\n\n"
|
|
77
|
+
f"Verification evidence:\n{report}"
|
|
78
|
+
),
|
|
79
|
+
benchmark,
|
|
80
|
+
output,
|
|
81
|
+
)
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
"""Evaluation prompts package."""
|
|
2
|
+
|
|
3
|
+
from framework.evaluation.prompts.base import (
|
|
4
|
+
JSON_RESPONSE_SCHEMA,
|
|
5
|
+
build_llm_judge_prompt,
|
|
6
|
+
)
|
|
7
|
+
from framework.evaluation.prompts.constraint import build_constraint_prompt
|
|
8
|
+
from framework.evaluation.prompts.planning import build_planning_prompt
|
|
9
|
+
from framework.evaluation.prompts.personalization import build_personalization_prompt
|
|
10
|
+
from framework.evaluation.prompts.adaptability import build_adaptability_prompt
|
|
11
|
+
from framework.evaluation.prompts.information_accuracy import build_information_accuracy_prompt
|