aeval-framework 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- aeval_framework-0.1.0.dist-info/METADATA +42 -0
- aeval_framework-0.1.0.dist-info/RECORD +63 -0
- aeval_framework-0.1.0.dist-info/WHEEL +4 -0
- aeval_framework-0.1.0.dist-info/entry_points.txt +2 -0
- agent_eval/__init__.py +14 -0
- agent_eval/api/__init__.py +14 -0
- agent_eval/api/app.py +82 -0
- agent_eval/api/events.py +96 -0
- agent_eval/api/routes/__init__.py +0 -0
- agent_eval/api/routes/datasets.py +441 -0
- agent_eval/api/routes/graders.py +19 -0
- agent_eval/api/routes/metrics.py +49 -0
- agent_eval/api/routes/runs.py +573 -0
- agent_eval/api/routes/suites.py +84 -0
- agent_eval/api/routes/tasks.py +114 -0
- agent_eval/api/standalone.py +105 -0
- agent_eval/cli.py +455 -0
- agent_eval/core/__init__.py +48 -0
- agent_eval/core/contract.py +296 -0
- agent_eval/core/metrics.py +184 -0
- agent_eval/core/runner.py +868 -0
- agent_eval/core/suite.py +60 -0
- agent_eval/core/types.py +227 -0
- agent_eval/dataset/__init__.py +31 -0
- agent_eval/dataset/models.py +199 -0
- agent_eval/dataset/quality.py +194 -0
- agent_eval/dataset/sources/__init__.py +45 -0
- agent_eval/dataset/sources/llm_generator.py +219 -0
- agent_eval/dataset/sources/manual.py +172 -0
- agent_eval/dataset/sources/regression.py +201 -0
- agent_eval/dataset/sources/trace_mining.py +277 -0
- agent_eval/dataset/storage.py +342 -0
- agent_eval/dataset/version.py +72 -0
- agent_eval/examples/__init__.py +0 -0
- agent_eval/examples/basic_usage.py +175 -0
- agent_eval/examples/mock_runner.py +195 -0
- agent_eval/graders/__init__.py +91 -0
- agent_eval/graders/artifact_check.py +114 -0
- agent_eval/graders/code_based.py +101 -0
- agent_eval/graders/human.py +77 -0
- agent_eval/graders/metric.py +142 -0
- agent_eval/graders/model_based.py +179 -0
- agent_eval/graders/state_check.py +106 -0
- agent_eval/graders/step_level.py +116 -0
- agent_eval/graders/tool_calls.py +102 -0
- agent_eval/graders/transcript.py +86 -0
- agent_eval/metrics/__init__.py +110 -0
- agent_eval/metrics/answer_relevancy.py +57 -0
- agent_eval/metrics/base.py +155 -0
- agent_eval/metrics/batch_evaluation.py +267 -0
- agent_eval/metrics/context_precision.py +62 -0
- agent_eval/metrics/context_recall.py +71 -0
- agent_eval/metrics/faithfulness.py +72 -0
- agent_eval/metrics/llm_judge.py +100 -0
- agent_eval/metrics/prompt_metric.py +150 -0
- agent_eval/metrics/pytest_plugin.py +308 -0
- agent_eval/metrics/report.py +149 -0
- agent_eval/metrics/synthetic_data.py +203 -0
- agent_eval/storage/__init__.py +17 -0
- agent_eval/storage/memory.py +95 -0
- agent_eval/storage/sqlite.py +240 -0
- agent_eval/trace/__init__.py +16 -0
- agent_eval/trace/phoenix.py +144 -0
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Metric grader — the built-in dispatcher from grader configs to Metric instances.
|
|
3
|
+
|
|
4
|
+
Task grader configs with `type: metric` route through this grader (D1):
|
|
5
|
+
- `name: metric` + `config.metric_name` (single dispatcher per task), or
|
|
6
|
+
- `name: <metric_name>` + `type: metric` (several metrics per task — the
|
|
7
|
+
runner falls back to this dispatcher for unregistered metric-type configs)
|
|
8
|
+
|
|
9
|
+
The metric registry and LLM function are injected by EvalRunner
|
|
10
|
+
(metrics_registry=..., llm_fn=...). Unregistered metric names score 0 with
|
|
11
|
+
an explicit reason; missing LLM configuration surfaces a clear config error
|
|
12
|
+
result instead of crashing the run.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
from agent_eval.core.contract import EvalContext
|
|
20
|
+
from agent_eval.core.types import (
|
|
21
|
+
EvalTask,
|
|
22
|
+
GraderConfig,
|
|
23
|
+
GraderResult,
|
|
24
|
+
GraderType,
|
|
25
|
+
TrialResult,
|
|
26
|
+
)
|
|
27
|
+
from agent_eval.metrics.base import Metric, MetricError
|
|
28
|
+
from agent_eval.metrics.llm_judge import LLMFn, LLMJudgeError, LLMNotConfiguredError
|
|
29
|
+
|
|
30
|
+
# 计算失败 (配置/解析/调用) 映射为 0 分结果, 不 crash run (D2)
|
|
31
|
+
_CALC_ERRORS = (LLMNotConfiguredError, LLMJudgeError, MetricError)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class MetricGrader:
|
|
35
|
+
"""按 config.metric_name 从注入注册表分发到对应 Metric 计算"""
|
|
36
|
+
|
|
37
|
+
name = "metric"
|
|
38
|
+
|
|
39
|
+
def __init__(
|
|
40
|
+
self,
|
|
41
|
+
metrics_registry: dict[str, Metric] | None = None,
|
|
42
|
+
llm_fn: LLMFn | None = None,
|
|
43
|
+
):
|
|
44
|
+
# 由 EvalRunner 组合根覆盖注入 (与 storage 注入同一模式)
|
|
45
|
+
self.metrics_registry: dict[str, Metric] = dict(metrics_registry or {})
|
|
46
|
+
self.llm_fn = llm_fn
|
|
47
|
+
|
|
48
|
+
async def grade(
|
|
49
|
+
self,
|
|
50
|
+
trial: TrialResult,
|
|
51
|
+
spans: list[dict[str, Any]],
|
|
52
|
+
task: EvalTask,
|
|
53
|
+
context: EvalContext | None = None,
|
|
54
|
+
) -> GraderResult:
|
|
55
|
+
config = self._active_config(task, context)
|
|
56
|
+
metric_name = self._metric_name(config)
|
|
57
|
+
|
|
58
|
+
if not metric_name:
|
|
59
|
+
return self._result(config, 0.0, False, "metric grader 缺少 config.metric_name")
|
|
60
|
+
|
|
61
|
+
metric = self.metrics_registry.get(metric_name)
|
|
62
|
+
if metric is None:
|
|
63
|
+
return self._result(
|
|
64
|
+
config, 0.0, False, f"未知指标: {metric_name} (未在 metrics_registry 注册)"
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
kwargs = self._measure_kwargs(config, trial)
|
|
68
|
+
try:
|
|
69
|
+
result = await metric.measure(**kwargs)
|
|
70
|
+
except _CALC_ERRORS as e:
|
|
71
|
+
return self._result(config, 0.0, False, f"配置/计算错误: {e}")
|
|
72
|
+
except Exception as e: # noqa: BLE001 — 指标实现方错误同样不 crash run
|
|
73
|
+
return self._result(config, 0.0, False, f"指标计算异常: {e}")
|
|
74
|
+
|
|
75
|
+
threshold = float(config.config.get("threshold", metric.threshold))
|
|
76
|
+
return self._result(
|
|
77
|
+
config,
|
|
78
|
+
result.score,
|
|
79
|
+
result.score >= threshold,
|
|
80
|
+
result.reason,
|
|
81
|
+
details={
|
|
82
|
+
**result.details,
|
|
83
|
+
"metric": result.name,
|
|
84
|
+
"metric_threshold": result.threshold,
|
|
85
|
+
"grader_threshold": threshold,
|
|
86
|
+
},
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
# ── Config resolution ────────────────────────────────────────────────
|
|
90
|
+
|
|
91
|
+
@staticmethod
|
|
92
|
+
def _active_config(task: EvalTask, context: EvalContext | None) -> GraderConfig:
|
|
93
|
+
"""当前生效的 metric 配置: runner 传入优先, 否则取名为 metric 的配置"""
|
|
94
|
+
if context is not None and context.grader_config is not None:
|
|
95
|
+
return context.grader_config
|
|
96
|
+
for g in task.graders:
|
|
97
|
+
if g.type == GraderType.METRIC:
|
|
98
|
+
return g
|
|
99
|
+
raise ValueError(f"task '{task.id}' has no metric grader config")
|
|
100
|
+
|
|
101
|
+
@staticmethod
|
|
102
|
+
def _metric_name(config: GraderConfig) -> str:
|
|
103
|
+
"""metric_name 显式配置优先; 命名分发型 (name=指标名) 取配置名"""
|
|
104
|
+
explicit = config.config.get("metric_name")
|
|
105
|
+
if explicit:
|
|
106
|
+
return str(explicit)
|
|
107
|
+
if config.name != MetricGrader.name:
|
|
108
|
+
return config.name
|
|
109
|
+
return ""
|
|
110
|
+
|
|
111
|
+
@staticmethod
|
|
112
|
+
def _measure_kwargs(config: GraderConfig, trial: TrialResult) -> dict[str, Any]:
|
|
113
|
+
"""从 trial transcript 与 grader config 提取 measure() 入参"""
|
|
114
|
+
first = trial.transcript[0] if trial.transcript else {}
|
|
115
|
+
last = trial.transcript[-1] if trial.transcript else {}
|
|
116
|
+
prompt = first.get("content", "") if isinstance(first, dict) else ""
|
|
117
|
+
output = last.get("content", "") if isinstance(last, dict) else ""
|
|
118
|
+
|
|
119
|
+
return {
|
|
120
|
+
"input": prompt,
|
|
121
|
+
"actual_output": output,
|
|
122
|
+
"expected_output": config.config.get("expected_output"),
|
|
123
|
+
"context": config.config.get("context"),
|
|
124
|
+
"retrieval_context": config.config.get("retrieval_context"),
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
@staticmethod
|
|
128
|
+
def _result(
|
|
129
|
+
config: GraderConfig,
|
|
130
|
+
score: float,
|
|
131
|
+
passed: bool,
|
|
132
|
+
explanation: str,
|
|
133
|
+
details: dict[str, Any] | None = None,
|
|
134
|
+
) -> GraderResult:
|
|
135
|
+
return GraderResult(
|
|
136
|
+
grader_name=config.name,
|
|
137
|
+
grader_type=GraderType.METRIC,
|
|
138
|
+
score=max(0.0, min(1.0, score)),
|
|
139
|
+
passed=passed,
|
|
140
|
+
explanation=explanation,
|
|
141
|
+
details=details or {},
|
|
142
|
+
)
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Model-based grader — LLM-as-Judge scoring.
|
|
3
|
+
|
|
4
|
+
Sends the trial transcript and a rubric to an LLM, which returns
|
|
5
|
+
a score between 0 and 1.
|
|
6
|
+
|
|
7
|
+
Config schema:
|
|
8
|
+
{
|
|
9
|
+
"rubric": "The response must contain...",
|
|
10
|
+
"dimensions": ["correctness", "completeness"],
|
|
11
|
+
"threshold": 0.7,
|
|
12
|
+
"model": "gpt-4o-mini", # optional
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
Requires either:
|
|
16
|
+
- A configured llm_fn callback, or
|
|
17
|
+
- An API key in the environment (OPENAI_API_KEY, etc.)
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import contextlib
|
|
23
|
+
import json
|
|
24
|
+
import os
|
|
25
|
+
from collections.abc import Callable
|
|
26
|
+
from typing import Any
|
|
27
|
+
|
|
28
|
+
from agent_eval.core.contract import EvalContext
|
|
29
|
+
from agent_eval.core.types import EvalTask, GraderResult, GraderType, TrialResult
|
|
30
|
+
|
|
31
|
+
# Type alias for LLM function: (system_prompt, user_message) -> str
|
|
32
|
+
LLMFn = Callable[[str, str], str]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class ModelBasedGrader:
|
|
36
|
+
"""LLM-as-Judge 评分器"""
|
|
37
|
+
|
|
38
|
+
name = "model_based"
|
|
39
|
+
|
|
40
|
+
def __init__(self, llm_fn: LLMFn | None = None):
|
|
41
|
+
"""
|
|
42
|
+
Args:
|
|
43
|
+
llm_fn: (system_prompt, user_message) → response text.
|
|
44
|
+
If None, uses OpenAI API with OPENAI_API_KEY.
|
|
45
|
+
"""
|
|
46
|
+
self._llm_fn = llm_fn
|
|
47
|
+
|
|
48
|
+
async def grade(
|
|
49
|
+
self,
|
|
50
|
+
trial: TrialResult,
|
|
51
|
+
spans: list[dict[str, Any]],
|
|
52
|
+
task: EvalTask,
|
|
53
|
+
context: EvalContext | None = None,
|
|
54
|
+
) -> GraderResult:
|
|
55
|
+
config = task.get_grader_config(self.name)
|
|
56
|
+
rubric = config.get("rubric", "")
|
|
57
|
+
dimensions = config.get("dimensions", ["quality"])
|
|
58
|
+
threshold = config.get("threshold", 0.7)
|
|
59
|
+
|
|
60
|
+
# Build judge prompt
|
|
61
|
+
prompt = self._build_prompt(trial, rubric, dimensions)
|
|
62
|
+
|
|
63
|
+
# Call LLM
|
|
64
|
+
llm_fn = self._llm_fn or self._default_llm_fn(config)
|
|
65
|
+
try:
|
|
66
|
+
raw = llm_fn("You are an evaluation expert.", prompt)
|
|
67
|
+
except Exception as e:
|
|
68
|
+
return GraderResult(
|
|
69
|
+
grader_name=self.name,
|
|
70
|
+
grader_type=GraderType.MODEL,
|
|
71
|
+
score=0.0,
|
|
72
|
+
passed=False,
|
|
73
|
+
explanation=f"LLM call failed: {e}",
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
# Parse scores
|
|
77
|
+
scores = self._parse_scores(raw, dimensions)
|
|
78
|
+
avg_score = sum(scores.values()) / len(scores) if scores else 0.0
|
|
79
|
+
|
|
80
|
+
return GraderResult(
|
|
81
|
+
grader_name=self.name,
|
|
82
|
+
grader_type=GraderType.MODEL,
|
|
83
|
+
score=avg_score,
|
|
84
|
+
passed=avg_score >= threshold,
|
|
85
|
+
explanation=f"LLM Judge scores: {scores}",
|
|
86
|
+
details={"dimensions": scores, "raw_response": raw},
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
def _build_prompt(
|
|
90
|
+
self,
|
|
91
|
+
trial: TrialResult,
|
|
92
|
+
rubric: str,
|
|
93
|
+
dimensions: list[str],
|
|
94
|
+
) -> str:
|
|
95
|
+
"""构建 judge prompt"""
|
|
96
|
+
input_msg = trial.transcript[0] if trial.transcript else "N/A"
|
|
97
|
+
output_msg = trial.transcript[-1] if trial.transcript else "N/A"
|
|
98
|
+
|
|
99
|
+
# 提取工具调用摘要
|
|
100
|
+
tools_used = list(set(
|
|
101
|
+
msg.get("tool_name", "")
|
|
102
|
+
for msg in trial.transcript
|
|
103
|
+
if msg.get("role") == "tool_call"
|
|
104
|
+
))
|
|
105
|
+
|
|
106
|
+
dims_json = ", ".join(f'"{d}": 0.0' for d in dimensions)
|
|
107
|
+
|
|
108
|
+
return f"""请根据以下评分标准对 Agent 表现进行评分。
|
|
109
|
+
|
|
110
|
+
## 评分标准
|
|
111
|
+
{rubric}
|
|
112
|
+
|
|
113
|
+
## 评分维度
|
|
114
|
+
{", ".join(dimensions)}
|
|
115
|
+
|
|
116
|
+
## Agent 执行记录
|
|
117
|
+
- 输入: {input_msg}
|
|
118
|
+
- 输出: {output_msg}
|
|
119
|
+
- 使用的工具: {tools_used or "无"}
|
|
120
|
+
|
|
121
|
+
请以 JSON 格式返回各维度评分 (0.0-1.0):
|
|
122
|
+
```json
|
|
123
|
+
{{{dims_json}}}
|
|
124
|
+
```"""
|
|
125
|
+
|
|
126
|
+
def _parse_scores(self, raw: str, dimensions: list[str]) -> dict[str, float]:
|
|
127
|
+
"""从 LLM 响应中解析分数"""
|
|
128
|
+
scores: dict[str, float] = {}
|
|
129
|
+
|
|
130
|
+
# 尝试提取 JSON
|
|
131
|
+
try:
|
|
132
|
+
# 查找 JSON 块
|
|
133
|
+
json_start = raw.find("{")
|
|
134
|
+
json_end = raw.rfind("}") + 1
|
|
135
|
+
if json_start >= 0 and json_end > json_start:
|
|
136
|
+
json_str = raw[json_start:json_end]
|
|
137
|
+
parsed = json.loads(json_str)
|
|
138
|
+
for dim in dimensions:
|
|
139
|
+
if dim in parsed:
|
|
140
|
+
with contextlib.suppress(ValueError, TypeError):
|
|
141
|
+
scores[dim] = float(parsed[dim])
|
|
142
|
+
except json.JSONDecodeError:
|
|
143
|
+
pass
|
|
144
|
+
|
|
145
|
+
# 如果解析失败,给所有维度默认分
|
|
146
|
+
if not scores:
|
|
147
|
+
for dim in dimensions:
|
|
148
|
+
scores[dim] = 0.5
|
|
149
|
+
|
|
150
|
+
return scores
|
|
151
|
+
|
|
152
|
+
def _default_llm_fn(self, config: dict[str, Any]) -> LLMFn:
|
|
153
|
+
"""创建默认的 LLM 调用函数"""
|
|
154
|
+
model = config.get("model", "gpt-4o-mini")
|
|
155
|
+
|
|
156
|
+
def call_llm(system: str, user: str) -> str:
|
|
157
|
+
try:
|
|
158
|
+
from openai import OpenAI
|
|
159
|
+
|
|
160
|
+
client = OpenAI(api_key=os.getenv("OPENAI_API_KEY"))
|
|
161
|
+
resp = client.chat.completions.create(
|
|
162
|
+
model=model,
|
|
163
|
+
messages=[
|
|
164
|
+
{"role": "system", "content": system},
|
|
165
|
+
{"role": "user", "content": user},
|
|
166
|
+
],
|
|
167
|
+
temperature=0.0,
|
|
168
|
+
max_tokens=500,
|
|
169
|
+
)
|
|
170
|
+
return resp.choices[0].message.content or ""
|
|
171
|
+
except ImportError as e:
|
|
172
|
+
raise RuntimeError(
|
|
173
|
+
"openai package not installed. "
|
|
174
|
+
"Install with: pip install openai"
|
|
175
|
+
) from e
|
|
176
|
+
except Exception as e:
|
|
177
|
+
raise RuntimeError(f"OpenAI API call failed: {e}") from e
|
|
178
|
+
|
|
179
|
+
return call_llm
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"""
|
|
2
|
+
State-check grader — environment state verification.
|
|
3
|
+
|
|
4
|
+
Validates the environment state after a trial:
|
|
5
|
+
- file_exists: file is present
|
|
6
|
+
- file_contains: file content contains a substring
|
|
7
|
+
- db_record: database record matches criteria
|
|
8
|
+
- custom: custom check function
|
|
9
|
+
|
|
10
|
+
Config schema:
|
|
11
|
+
{
|
|
12
|
+
"expectations": [
|
|
13
|
+
{"type": "file_exists", "path": "output.py"},
|
|
14
|
+
{"type": "file_contains", "path": "output.py", "value": "def main"},
|
|
15
|
+
{"type": "db_record", "table": "users", "match": {"id": 1}},
|
|
16
|
+
],
|
|
17
|
+
"threshold": 1.0
|
|
18
|
+
}
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import re
|
|
24
|
+
from typing import Any
|
|
25
|
+
|
|
26
|
+
from agent_eval.core.contract import EvalContext
|
|
27
|
+
from agent_eval.core.types import EvalTask, GraderResult, GraderType, TrialResult
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class StateCheckGrader:
|
|
31
|
+
"""环境状态检查评分器"""
|
|
32
|
+
|
|
33
|
+
name = "state_check"
|
|
34
|
+
|
|
35
|
+
async def grade(
|
|
36
|
+
self,
|
|
37
|
+
trial: TrialResult,
|
|
38
|
+
spans: list[dict[str, Any]],
|
|
39
|
+
task: EvalTask,
|
|
40
|
+
context: EvalContext | None = None,
|
|
41
|
+
) -> GraderResult:
|
|
42
|
+
config = task.get_grader_config(self.name)
|
|
43
|
+
expectations = config.get("expectations", [])
|
|
44
|
+
threshold = config.get("threshold", 1.0)
|
|
45
|
+
|
|
46
|
+
if not expectations:
|
|
47
|
+
return GraderResult(
|
|
48
|
+
grader_name=self.name,
|
|
49
|
+
grader_type=GraderType.STATE,
|
|
50
|
+
score=1.0,
|
|
51
|
+
passed=True,
|
|
52
|
+
explanation="No expectations configured, auto-pass",
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
passed_count = 0
|
|
56
|
+
details: list[dict[str, Any]] = []
|
|
57
|
+
|
|
58
|
+
for exp in expectations:
|
|
59
|
+
exp_type = exp.get("type", "file_exists")
|
|
60
|
+
ok = False
|
|
61
|
+
|
|
62
|
+
if exp_type == "file_exists":
|
|
63
|
+
files = trial.outcome.get("files", {})
|
|
64
|
+
ok = exp["path"] in files
|
|
65
|
+
|
|
66
|
+
elif exp_type == "file_contains":
|
|
67
|
+
files = trial.outcome.get("files", {})
|
|
68
|
+
content = files.get(exp["path"], "")
|
|
69
|
+
ok = exp["value"] in content
|
|
70
|
+
|
|
71
|
+
elif exp_type == "file_regex":
|
|
72
|
+
files = trial.outcome.get("files", {})
|
|
73
|
+
content = files.get(exp["path"], "")
|
|
74
|
+
ok = bool(re.search(exp["value"], content))
|
|
75
|
+
|
|
76
|
+
elif exp_type == "db_record":
|
|
77
|
+
records = trial.outcome.get("db_records", [])
|
|
78
|
+
match_criteria = exp.get("match", {})
|
|
79
|
+
ok = any(
|
|
80
|
+
all(r.get(k) == v for k, v in match_criteria.items())
|
|
81
|
+
for r in records
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
elif exp_type == "no_conflict_markers":
|
|
85
|
+
files = trial.outcome.get("files", {})
|
|
86
|
+
content = files.get(exp["path"], "")
|
|
87
|
+
ok = "<<<<<<<" not in content and ">>>>>>>" not in content
|
|
88
|
+
|
|
89
|
+
else:
|
|
90
|
+
ok = False
|
|
91
|
+
|
|
92
|
+
if ok:
|
|
93
|
+
passed_count += 1
|
|
94
|
+
details.append({"expectation": exp, "passed": ok})
|
|
95
|
+
|
|
96
|
+
total = len(expectations)
|
|
97
|
+
score = passed_count / total if total > 0 else 1.0
|
|
98
|
+
|
|
99
|
+
return GraderResult(
|
|
100
|
+
grader_name=self.name,
|
|
101
|
+
grader_type=GraderType.STATE,
|
|
102
|
+
score=score,
|
|
103
|
+
passed=score >= threshold,
|
|
104
|
+
explanation=f"{passed_count}/{total} state checks passed",
|
|
105
|
+
details={"expectations": details},
|
|
106
|
+
)
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Step-level grader — compares the agent's tool-call sequence against an
|
|
3
|
+
expected trace (design decision D6, first version: exact index-by-index
|
|
4
|
+
comparison only).
|
|
5
|
+
|
|
6
|
+
From the trace spans, extracts the sequence of ``tool.call`` steps (tool name
|
|
7
|
+
from span attributes, falling back to the span name), aligns it with the
|
|
8
|
+
task's ``expected_trace`` config by index, reports the first wrong step and
|
|
9
|
+
scores ``correct_steps / total_steps``.
|
|
10
|
+
|
|
11
|
+
Config schema:
|
|
12
|
+
{
|
|
13
|
+
"expected_trace": ["fs_read", "fs_write", "bash"], # required
|
|
14
|
+
"threshold": 0.7, # optional, pass threshold
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
When ``expected_trace`` is not configured the grader auto-passes (nothing to
|
|
18
|
+
compare against), mirroring code_based's behavior with no checks.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
from typing import Any
|
|
24
|
+
|
|
25
|
+
from agent_eval.core.contract import EvalContext
|
|
26
|
+
from agent_eval.core.types import EvalTask, GraderResult, GraderType, TrialResult
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class StepLevelGrader:
|
|
30
|
+
"""步骤级评估 — expected_trace 按索引对照, 定位首个错误步骤"""
|
|
31
|
+
|
|
32
|
+
name = "step_level"
|
|
33
|
+
|
|
34
|
+
async def grade(
|
|
35
|
+
self,
|
|
36
|
+
trial: TrialResult,
|
|
37
|
+
spans: list[dict[str, Any]],
|
|
38
|
+
task: EvalTask,
|
|
39
|
+
context: EvalContext | None = None,
|
|
40
|
+
) -> GraderResult:
|
|
41
|
+
config = task.get_grader_config(self.name)
|
|
42
|
+
expected = config.get("expected_trace")
|
|
43
|
+
threshold = config.get("threshold", 0.7)
|
|
44
|
+
|
|
45
|
+
actual = self._extract_steps(spans)
|
|
46
|
+
|
|
47
|
+
if not expected:
|
|
48
|
+
return GraderResult(
|
|
49
|
+
grader_name=self.name,
|
|
50
|
+
grader_type=GraderType.CUSTOM,
|
|
51
|
+
score=1.0,
|
|
52
|
+
passed=True,
|
|
53
|
+
explanation="No expected_trace configured, auto-pass",
|
|
54
|
+
details={"actual_steps": actual},
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
total = len(expected)
|
|
58
|
+
step_details: list[dict[str, Any]] = []
|
|
59
|
+
first_error: int | None = None
|
|
60
|
+
|
|
61
|
+
for i in range(total):
|
|
62
|
+
expected_step = expected[i]
|
|
63
|
+
actual_step = actual[i] if i < len(actual) else None
|
|
64
|
+
ok = actual_step == expected_step
|
|
65
|
+
if not ok and first_error is None:
|
|
66
|
+
first_error = i
|
|
67
|
+
step_details.append({
|
|
68
|
+
"index": i,
|
|
69
|
+
"expected": expected_step,
|
|
70
|
+
"actual": actual_step,
|
|
71
|
+
"correct": ok,
|
|
72
|
+
})
|
|
73
|
+
|
|
74
|
+
correct_count = sum(1 for s in step_details if s["correct"])
|
|
75
|
+
score = correct_count / total if total > 0 else 1.0
|
|
76
|
+
|
|
77
|
+
explanation = f"{correct_count}/{total} steps correct"
|
|
78
|
+
if first_error is not None:
|
|
79
|
+
explanation += (
|
|
80
|
+
f"; first error at step {first_error}: "
|
|
81
|
+
f"expected '{expected[first_error]}', "
|
|
82
|
+
f"got '{actual[first_error] if first_error < len(actual) else None}'"
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
return GraderResult(
|
|
86
|
+
grader_name=self.name,
|
|
87
|
+
grader_type=GraderType.CUSTOM,
|
|
88
|
+
score=score,
|
|
89
|
+
passed=score >= threshold,
|
|
90
|
+
explanation=explanation,
|
|
91
|
+
details={
|
|
92
|
+
"steps": step_details,
|
|
93
|
+
"first_error_step": first_error,
|
|
94
|
+
"actual_steps": actual,
|
|
95
|
+
"extra_steps": (
|
|
96
|
+
actual[total:] if len(actual) > total else []
|
|
97
|
+
),
|
|
98
|
+
},
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
@staticmethod
|
|
102
|
+
def _extract_steps(spans: list[dict[str, Any]]) -> list[str]:
|
|
103
|
+
"""从 spans 提取 tool.call 步骤序列 (工具名, 回退到 span 名称)"""
|
|
104
|
+
steps: list[str] = []
|
|
105
|
+
for span in spans:
|
|
106
|
+
name = span.get("name", "")
|
|
107
|
+
if "tool.call" not in name and "tool_call" not in name:
|
|
108
|
+
continue
|
|
109
|
+
attrs = span.get("attributes", {}) or {}
|
|
110
|
+
tool_name = (
|
|
111
|
+
attrs.get("agenthub.tool_name")
|
|
112
|
+
or attrs.get("tool_name")
|
|
113
|
+
or name
|
|
114
|
+
)
|
|
115
|
+
steps.append(str(tool_name))
|
|
116
|
+
return steps
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Tool-calls grader — tool call validation.
|
|
3
|
+
|
|
4
|
+
Validates that the agent:
|
|
5
|
+
- Called required tools
|
|
6
|
+
- Did not use forbidden tools
|
|
7
|
+
- (Optional) Called tools in a specific order
|
|
8
|
+
|
|
9
|
+
Config schema:
|
|
10
|
+
{
|
|
11
|
+
"required_tools": ["fs_read", "fs_write"],
|
|
12
|
+
"forbidden_tools": ["dangerous_tool"],
|
|
13
|
+
"threshold": 1.0
|
|
14
|
+
}
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from typing import Any
|
|
20
|
+
|
|
21
|
+
from agent_eval.core.contract import EvalContext
|
|
22
|
+
from agent_eval.core.types import EvalTask, GraderResult, GraderType, TrialResult
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class ToolCallsGrader:
|
|
26
|
+
"""工具调用验证评分器"""
|
|
27
|
+
|
|
28
|
+
name = "tool_calls"
|
|
29
|
+
|
|
30
|
+
async def grade(
|
|
31
|
+
self,
|
|
32
|
+
trial: TrialResult,
|
|
33
|
+
spans: list[dict[str, Any]],
|
|
34
|
+
task: EvalTask,
|
|
35
|
+
context: EvalContext | None = None,
|
|
36
|
+
) -> GraderResult:
|
|
37
|
+
config = task.get_grader_config(self.name)
|
|
38
|
+
required_tools = config.get("required_tools", [])
|
|
39
|
+
forbidden_tools = config.get("forbidden_tools", [])
|
|
40
|
+
threshold = config.get("threshold", 1.0)
|
|
41
|
+
|
|
42
|
+
# 从 spans 提取工具调用
|
|
43
|
+
tool_calls = self._extract_tool_calls(spans)
|
|
44
|
+
used_tools = [tc["name"] for tc in tool_calls]
|
|
45
|
+
|
|
46
|
+
# 检查必须使用的工具
|
|
47
|
+
missing = [t for t in required_tools if t not in used_tools]
|
|
48
|
+
# 检查禁止使用的工具
|
|
49
|
+
violated = [t for t in forbidden_tools if t in used_tools]
|
|
50
|
+
|
|
51
|
+
# 计算分数
|
|
52
|
+
if required_tools:
|
|
53
|
+
found = len(required_tools) - len(missing)
|
|
54
|
+
score = found / len(required_tools)
|
|
55
|
+
else:
|
|
56
|
+
score = 1.0
|
|
57
|
+
|
|
58
|
+
# 违反禁止工具则直接 0 分
|
|
59
|
+
if violated:
|
|
60
|
+
score = 0.0
|
|
61
|
+
|
|
62
|
+
passed = score >= threshold
|
|
63
|
+
|
|
64
|
+
# 构建解释
|
|
65
|
+
parts = []
|
|
66
|
+
if used_tools:
|
|
67
|
+
parts.append(f"Used: {used_tools}")
|
|
68
|
+
if missing:
|
|
69
|
+
parts.append(f"Missing required: {missing}")
|
|
70
|
+
if violated:
|
|
71
|
+
parts.append(f"Violated forbidden: {violated}")
|
|
72
|
+
explanation = "; ".join(parts) if parts else "No tool calls checked"
|
|
73
|
+
|
|
74
|
+
return GraderResult(
|
|
75
|
+
grader_name=self.name,
|
|
76
|
+
grader_type=GraderType.TOOL_CALLS,
|
|
77
|
+
score=score,
|
|
78
|
+
passed=passed,
|
|
79
|
+
explanation=explanation,
|
|
80
|
+
details={
|
|
81
|
+
"tool_calls": tool_calls,
|
|
82
|
+
"used_tools": used_tools,
|
|
83
|
+
"missing": missing,
|
|
84
|
+
"violated": violated,
|
|
85
|
+
},
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
def _extract_tool_calls(self, spans: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
89
|
+
"""从 spans 中提取工具调用"""
|
|
90
|
+
tool_calls = []
|
|
91
|
+
for span in spans:
|
|
92
|
+
name = span.get("name", "")
|
|
93
|
+
attrs = span.get("attributes", {})
|
|
94
|
+
|
|
95
|
+
if "tool.call" in name or "tool_call" in name:
|
|
96
|
+
tool_calls.append({
|
|
97
|
+
"name": attrs.get("agenthub.tool_name")
|
|
98
|
+
or attrs.get("tool.name", ""),
|
|
99
|
+
"success": attrs.get("agenthub.success", True),
|
|
100
|
+
})
|
|
101
|
+
|
|
102
|
+
return tool_calls
|