aeval-framework 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- aeval_framework-0.1.0.dist-info/METADATA +42 -0
- aeval_framework-0.1.0.dist-info/RECORD +63 -0
- aeval_framework-0.1.0.dist-info/WHEEL +4 -0
- aeval_framework-0.1.0.dist-info/entry_points.txt +2 -0
- agent_eval/__init__.py +14 -0
- agent_eval/api/__init__.py +14 -0
- agent_eval/api/app.py +82 -0
- agent_eval/api/events.py +96 -0
- agent_eval/api/routes/__init__.py +0 -0
- agent_eval/api/routes/datasets.py +441 -0
- agent_eval/api/routes/graders.py +19 -0
- agent_eval/api/routes/metrics.py +49 -0
- agent_eval/api/routes/runs.py +573 -0
- agent_eval/api/routes/suites.py +84 -0
- agent_eval/api/routes/tasks.py +114 -0
- agent_eval/api/standalone.py +105 -0
- agent_eval/cli.py +455 -0
- agent_eval/core/__init__.py +48 -0
- agent_eval/core/contract.py +296 -0
- agent_eval/core/metrics.py +184 -0
- agent_eval/core/runner.py +868 -0
- agent_eval/core/suite.py +60 -0
- agent_eval/core/types.py +227 -0
- agent_eval/dataset/__init__.py +31 -0
- agent_eval/dataset/models.py +199 -0
- agent_eval/dataset/quality.py +194 -0
- agent_eval/dataset/sources/__init__.py +45 -0
- agent_eval/dataset/sources/llm_generator.py +219 -0
- agent_eval/dataset/sources/manual.py +172 -0
- agent_eval/dataset/sources/regression.py +201 -0
- agent_eval/dataset/sources/trace_mining.py +277 -0
- agent_eval/dataset/storage.py +342 -0
- agent_eval/dataset/version.py +72 -0
- agent_eval/examples/__init__.py +0 -0
- agent_eval/examples/basic_usage.py +175 -0
- agent_eval/examples/mock_runner.py +195 -0
- agent_eval/graders/__init__.py +91 -0
- agent_eval/graders/artifact_check.py +114 -0
- agent_eval/graders/code_based.py +101 -0
- agent_eval/graders/human.py +77 -0
- agent_eval/graders/metric.py +142 -0
- agent_eval/graders/model_based.py +179 -0
- agent_eval/graders/state_check.py +106 -0
- agent_eval/graders/step_level.py +116 -0
- agent_eval/graders/tool_calls.py +102 -0
- agent_eval/graders/transcript.py +86 -0
- agent_eval/metrics/__init__.py +110 -0
- agent_eval/metrics/answer_relevancy.py +57 -0
- agent_eval/metrics/base.py +155 -0
- agent_eval/metrics/batch_evaluation.py +267 -0
- agent_eval/metrics/context_precision.py +62 -0
- agent_eval/metrics/context_recall.py +71 -0
- agent_eval/metrics/faithfulness.py +72 -0
- agent_eval/metrics/llm_judge.py +100 -0
- agent_eval/metrics/prompt_metric.py +150 -0
- agent_eval/metrics/pytest_plugin.py +308 -0
- agent_eval/metrics/report.py +149 -0
- agent_eval/metrics/synthetic_data.py +203 -0
- agent_eval/storage/__init__.py +17 -0
- agent_eval/storage/memory.py +95 -0
- agent_eval/storage/sqlite.py +240 -0
- agent_eval/trace/__init__.py +16 -0
- agent_eval/trace/phoenix.py +144 -0
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Mock AgentRunner for testing and demonstration.
|
|
3
|
+
|
|
4
|
+
Simulates an agent by returning predefined results, with per-task scripted
|
|
5
|
+
behaviors for exercising the framework's failure paths.
|
|
6
|
+
|
|
7
|
+
Usage:
|
|
8
|
+
from agent_eval.examples.mock_runner import MockAgentRunner
|
|
9
|
+
|
|
10
|
+
# Random behavior (demo)
|
|
11
|
+
runner = EvalRunner(agent_runner=MockAgentRunner())
|
|
12
|
+
|
|
13
|
+
# Scripted behavior (tests): each task consumes its behavior list in
|
|
14
|
+
# order across calls; the last entry repeats once exhausted.
|
|
15
|
+
agent = MockAgentRunner(
|
|
16
|
+
latency_range=(0.0, 0.01),
|
|
17
|
+
script={
|
|
18
|
+
"task_ok": ["success"],
|
|
19
|
+
"task_flaky": ["transient", "transient", "success"],
|
|
20
|
+
"task_dead": ["failure"],
|
|
21
|
+
"task_slow": ["timeout"],
|
|
22
|
+
},
|
|
23
|
+
)
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import asyncio
|
|
29
|
+
import random
|
|
30
|
+
import uuid
|
|
31
|
+
from typing import Any
|
|
32
|
+
|
|
33
|
+
from agent_eval.core.contract import TransientError
|
|
34
|
+
from agent_eval.core.types import EvalTask
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class MockAgentRunner:
|
|
38
|
+
"""
|
|
39
|
+
模拟 AgentRunner。
|
|
40
|
+
|
|
41
|
+
用于测试和演示框架功能,无需真实 Agent 系统。
|
|
42
|
+
支持脚本化场景: success / failure / transient / timeout。
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
def __init__(
|
|
46
|
+
self,
|
|
47
|
+
success_rate: float = 0.7,
|
|
48
|
+
latency_range: tuple[float, float] = (0.1, 0.5),
|
|
49
|
+
script: dict[str, list[str]] | None = None,
|
|
50
|
+
timeout_duration: float = 10.0,
|
|
51
|
+
):
|
|
52
|
+
"""
|
|
53
|
+
Args:
|
|
54
|
+
success_rate: 随机模式下的模拟成功率 (0.0-1.0)
|
|
55
|
+
latency_range: 模拟延迟范围 (秒)
|
|
56
|
+
script: task_id → 行为序列 ("success"|"failure"|"transient"|"timeout"),
|
|
57
|
+
逐次调用消耗, 耗尽后重复最后一项
|
|
58
|
+
timeout_duration: "timeout" 行为的挂起时长 (秒),
|
|
59
|
+
配合 EvalRunner(per_trial_timeout=...) 触发超时
|
|
60
|
+
"""
|
|
61
|
+
self.success_rate = success_rate
|
|
62
|
+
self.latency_range = latency_range
|
|
63
|
+
self.script = script or {}
|
|
64
|
+
self.timeout_duration = timeout_duration
|
|
65
|
+
self.call_counts: dict[str, int] = {}
|
|
66
|
+
|
|
67
|
+
def _next_behavior(self, task_id: str, call_index: int) -> str | None:
|
|
68
|
+
"""取该 task 指定调用的脚本行为 (无脚本返回 None = 随机模式)"""
|
|
69
|
+
behaviors = self.script.get(task_id)
|
|
70
|
+
if not behaviors:
|
|
71
|
+
return None
|
|
72
|
+
return behaviors[min(call_index, len(behaviors) - 1)]
|
|
73
|
+
|
|
74
|
+
async def run(
|
|
75
|
+
self,
|
|
76
|
+
task: EvalTask,
|
|
77
|
+
) -> tuple[str, list[dict[str, Any]], dict[str, Any]]:
|
|
78
|
+
"""
|
|
79
|
+
模拟 Agent 执行。
|
|
80
|
+
|
|
81
|
+
Returns:
|
|
82
|
+
trace_id, transcript, outcome
|
|
83
|
+
"""
|
|
84
|
+
index = self.call_counts.get(task.id, 0)
|
|
85
|
+
self.call_counts[task.id] = index + 1
|
|
86
|
+
behavior = self._next_behavior(task.id, index)
|
|
87
|
+
|
|
88
|
+
# 超时场景: 长时间挂起, 由框架的 per_trial_timeout 打断
|
|
89
|
+
if behavior == "timeout":
|
|
90
|
+
await asyncio.sleep(self.timeout_duration)
|
|
91
|
+
|
|
92
|
+
# 模拟延迟
|
|
93
|
+
latency = random.uniform(*self.latency_range)
|
|
94
|
+
await asyncio.sleep(latency)
|
|
95
|
+
|
|
96
|
+
# 瞬态错误场景: 框架按指数退避重试
|
|
97
|
+
if behavior == "transient":
|
|
98
|
+
raise TransientError(
|
|
99
|
+
f"Mock transient failure for {task.id} (call {index + 1})"
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
# 生成 trace_id (编码 task id, 便于 MockTraceProvider 关联 spans)
|
|
103
|
+
trace_id = f"trace_{task.id}_{uuid.uuid4().hex[:8]}"
|
|
104
|
+
|
|
105
|
+
# 构建 transcript
|
|
106
|
+
transcript = [
|
|
107
|
+
{
|
|
108
|
+
"role": "user",
|
|
109
|
+
"content": task.prompt,
|
|
110
|
+
},
|
|
111
|
+
{
|
|
112
|
+
"role": "assistant",
|
|
113
|
+
"content": f"Mock response for task: {task.id}",
|
|
114
|
+
},
|
|
115
|
+
]
|
|
116
|
+
|
|
117
|
+
# 构建 outcome (模拟成功/失败)
|
|
118
|
+
if behavior == "success":
|
|
119
|
+
success = True
|
|
120
|
+
elif behavior == "failure":
|
|
121
|
+
success = False
|
|
122
|
+
else:
|
|
123
|
+
success = random.random() < self.success_rate
|
|
124
|
+
|
|
125
|
+
outcome: dict[str, Any] = {
|
|
126
|
+
"success": success,
|
|
127
|
+
"files": {
|
|
128
|
+
"output.py": f"# Mock output for {task.id}\ndef hello(): pass\n",
|
|
129
|
+
},
|
|
130
|
+
"artifacts": [
|
|
131
|
+
{
|
|
132
|
+
"type": "code_file",
|
|
133
|
+
"id": f"art_{uuid.uuid4().hex[:8]}",
|
|
134
|
+
"content": f"# Generated code for {task.id}",
|
|
135
|
+
}
|
|
136
|
+
] if success else [],
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
return trace_id, transcript, outcome
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
class MockTraceProvider:
|
|
143
|
+
"""模拟 TraceProvider
|
|
144
|
+
|
|
145
|
+
可选按 task id 关联 span 数据: trace_id 形如 "trace_{task_id}_{suffix}"
|
|
146
|
+
时返回 spans_by_task[task_id] (若已配置), 否则返回默认 spans。
|
|
147
|
+
"""
|
|
148
|
+
|
|
149
|
+
def __init__(
|
|
150
|
+
self,
|
|
151
|
+
spans_by_task: dict[str, list[dict[str, Any]]] | None = None,
|
|
152
|
+
default_spans: list[dict[str, Any]] | None = None,
|
|
153
|
+
):
|
|
154
|
+
self.spans_by_task = spans_by_task or {}
|
|
155
|
+
self.default_spans = default_spans or self._build_default_spans()
|
|
156
|
+
|
|
157
|
+
@staticmethod
|
|
158
|
+
def _build_default_spans() -> list[dict[str, Any]]:
|
|
159
|
+
return [
|
|
160
|
+
{
|
|
161
|
+
"name": "agent.turn",
|
|
162
|
+
"attributes": {
|
|
163
|
+
"agenthub.total_tokens": 150,
|
|
164
|
+
},
|
|
165
|
+
"start_time": "2026-08-29T10:00:00Z",
|
|
166
|
+
"end_time": "2026-08-29T10:00:01Z",
|
|
167
|
+
"status": {"status_code": "OK"},
|
|
168
|
+
},
|
|
169
|
+
{
|
|
170
|
+
"name": "tool.call",
|
|
171
|
+
"attributes": {
|
|
172
|
+
"agenthub.tool_name": "fs_write",
|
|
173
|
+
"agenthub.success": True,
|
|
174
|
+
},
|
|
175
|
+
"start_time": "2026-08-29T10:00:01Z",
|
|
176
|
+
"end_time": "2026-08-29T10:00:02Z",
|
|
177
|
+
"status": {"status_code": "OK"},
|
|
178
|
+
},
|
|
179
|
+
]
|
|
180
|
+
|
|
181
|
+
async def get_spans(self, trace_id: str) -> list[dict[str, Any]]:
|
|
182
|
+
"""返回模拟 span 数据 (优先按 task id 匹配)"""
|
|
183
|
+
if trace_id.startswith("trace_"):
|
|
184
|
+
remainder = trace_id[len("trace_"):]
|
|
185
|
+
for task_id, spans in self.spans_by_task.items():
|
|
186
|
+
if remainder == task_id or remainder.startswith(f"{task_id}_"):
|
|
187
|
+
return spans
|
|
188
|
+
return self.default_spans
|
|
189
|
+
|
|
190
|
+
async def get_trace_ids(
|
|
191
|
+
self,
|
|
192
|
+
filters: dict[str, Any] | None = None,
|
|
193
|
+
limit: int = 100,
|
|
194
|
+
) -> list[str]:
|
|
195
|
+
return [f"trace_mock_{i}" for i in range(min(limit, 5))]
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Built-in graders for the Aeval evaluation framework.
|
|
3
|
+
|
|
4
|
+
Provides 9 built-in graders covering agent eval scenarios:
|
|
5
|
+
- code_based: Deterministic checks (string/regex matching)
|
|
6
|
+
- model_based: LLM-as-Judge
|
|
7
|
+
- state_check: Environment state verification
|
|
8
|
+
- tool_calls: Tool call validation
|
|
9
|
+
- transcript: Transcript analysis (turns/tokens/redundancy)
|
|
10
|
+
- artifact_check: Artifact verification
|
|
11
|
+
- human: Human expert scoring (pending semantics, async score submission)
|
|
12
|
+
- step_level: Step-level evaluation (expected_trace comparison)
|
|
13
|
+
- metric: LLM 输出质量指标分发 (按 config.metric_name 路由到 metrics 注册表)
|
|
14
|
+
|
|
15
|
+
Usage:
|
|
16
|
+
from agent_eval.graders import DEFAULT_GRADERS, get_grader_catalog
|
|
17
|
+
|
|
18
|
+
runner = EvalRunner(
|
|
19
|
+
agent_runner=my_runner,
|
|
20
|
+
graders=DEFAULT_GRADERS, # Use all built-in graders
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
# API listing (name/type/description)
|
|
24
|
+
catalog = get_grader_catalog()
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from typing import Any
|
|
28
|
+
|
|
29
|
+
from agent_eval.core.types import GraderType
|
|
30
|
+
from agent_eval.graders.artifact_check import ArtifactCheckGrader
|
|
31
|
+
from agent_eval.graders.code_based import CodeBasedGrader
|
|
32
|
+
from agent_eval.graders.human import HumanGrader
|
|
33
|
+
from agent_eval.graders.metric import MetricGrader
|
|
34
|
+
from agent_eval.graders.model_based import ModelBasedGrader
|
|
35
|
+
from agent_eval.graders.state_check import StateCheckGrader
|
|
36
|
+
from agent_eval.graders.step_level import StepLevelGrader
|
|
37
|
+
from agent_eval.graders.tool_calls import ToolCallsGrader
|
|
38
|
+
from agent_eval.graders.transcript import TranscriptGrader
|
|
39
|
+
|
|
40
|
+
# 注册表: name → {grader 实例, 类型, 描述} (供 API 列举与默认装配)
|
|
41
|
+
GRADER_REGISTRY: dict[str, dict[str, Any]] = {}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _register(grader: Any, grader_type: GraderType, description: str) -> None:
|
|
45
|
+
GRADER_REGISTRY[grader.name] = {
|
|
46
|
+
"grader": grader,
|
|
47
|
+
"type": grader_type,
|
|
48
|
+
"description": description,
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
_register(CodeBasedGrader(), GraderType.CODE, "确定性评分:字符串/正则/精确匹配检查")
|
|
53
|
+
_register(ModelBasedGrader(), GraderType.MODEL, "LLM-as-Judge 评分")
|
|
54
|
+
_register(StateCheckGrader(), GraderType.STATE, "环境状态检查")
|
|
55
|
+
_register(ToolCallsGrader(), GraderType.TOOL_CALLS, "工具调用验证 (必须/禁止调用)")
|
|
56
|
+
_register(TranscriptGrader(), GraderType.TRANSCRIPT, "转录记录分析 (轮次/Token 冗余)")
|
|
57
|
+
_register(ArtifactCheckGrader(), GraderType.ARTIFACT, "产物检查 (类型/内容正则)")
|
|
58
|
+
_register(HumanGrader(), GraderType.CUSTOM, "人工评分 (pending 语义, 异步回传)")
|
|
59
|
+
_register(StepLevelGrader(), GraderType.CUSTOM, "步骤级评估 (expected_trace 对照)")
|
|
60
|
+
_register(MetricGrader(), GraderType.METRIC, "LLM 输出质量指标 (按 metric_name 从注册表分发)")
|
|
61
|
+
|
|
62
|
+
# 默认内置 grader 实例列表
|
|
63
|
+
DEFAULT_GRADERS = [entry["grader"] for entry in GRADER_REGISTRY.values()]
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def get_grader_catalog() -> list[dict[str, str]]:
|
|
67
|
+
"""列出可用 grader (name/type/description), 供 GET /graders 使用"""
|
|
68
|
+
return [
|
|
69
|
+
{
|
|
70
|
+
"name": name,
|
|
71
|
+
"type": entry["type"].value,
|
|
72
|
+
"description": entry["description"],
|
|
73
|
+
}
|
|
74
|
+
for name, entry in GRADER_REGISTRY.items()
|
|
75
|
+
]
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
__all__ = [
|
|
79
|
+
"DEFAULT_GRADERS",
|
|
80
|
+
"GRADER_REGISTRY",
|
|
81
|
+
"get_grader_catalog",
|
|
82
|
+
"CodeBasedGrader",
|
|
83
|
+
"ModelBasedGrader",
|
|
84
|
+
"StateCheckGrader",
|
|
85
|
+
"ToolCallsGrader",
|
|
86
|
+
"TranscriptGrader",
|
|
87
|
+
"ArtifactCheckGrader",
|
|
88
|
+
"HumanGrader",
|
|
89
|
+
"StepLevelGrader",
|
|
90
|
+
"MetricGrader",
|
|
91
|
+
]
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Artifact-check grader — artifact verification.
|
|
3
|
+
|
|
4
|
+
Validates artifacts produced by the agent:
|
|
5
|
+
- Artifact exists
|
|
6
|
+
- Artifact type matches expected
|
|
7
|
+
- Artifact content matches regex pattern
|
|
8
|
+
|
|
9
|
+
Config schema:
|
|
10
|
+
{
|
|
11
|
+
"expected_type": "code_file",
|
|
12
|
+
"content_regex": "def \\w+\\(",
|
|
13
|
+
"threshold": 1.0
|
|
14
|
+
}
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import re
|
|
20
|
+
from typing import Any
|
|
21
|
+
|
|
22
|
+
from agent_eval.core.contract import EvalContext
|
|
23
|
+
from agent_eval.core.types import EvalTask, GraderResult, GraderType, TrialResult
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class ArtifactCheckGrader:
|
|
27
|
+
"""产物检查评分器"""
|
|
28
|
+
|
|
29
|
+
name = "artifact_check"
|
|
30
|
+
|
|
31
|
+
async def grade(
|
|
32
|
+
self,
|
|
33
|
+
trial: TrialResult,
|
|
34
|
+
spans: list[dict[str, Any]],
|
|
35
|
+
task: EvalTask,
|
|
36
|
+
context: EvalContext | None = None,
|
|
37
|
+
) -> GraderResult:
|
|
38
|
+
config = task.get_grader_config(self.name)
|
|
39
|
+
expected_type = config.get("expected_type")
|
|
40
|
+
content_regex = config.get("content_regex")
|
|
41
|
+
threshold = config.get("threshold", 1.0)
|
|
42
|
+
|
|
43
|
+
# 从 outcome 或 spans 提取产物
|
|
44
|
+
artifacts = self._extract_artifacts(trial, spans)
|
|
45
|
+
|
|
46
|
+
if not artifacts:
|
|
47
|
+
return GraderResult(
|
|
48
|
+
grader_name=self.name,
|
|
49
|
+
grader_type=GraderType.ARTIFACT,
|
|
50
|
+
score=0.0,
|
|
51
|
+
passed=False,
|
|
52
|
+
explanation="No artifacts produced",
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
# 检查类型
|
|
56
|
+
if expected_type:
|
|
57
|
+
types = [a.get("type", "") for a in artifacts]
|
|
58
|
+
if expected_type not in types:
|
|
59
|
+
return GraderResult(
|
|
60
|
+
grader_name=self.name,
|
|
61
|
+
grader_type=GraderType.ARTIFACT,
|
|
62
|
+
score=0.0,
|
|
63
|
+
passed=False,
|
|
64
|
+
explanation=(
|
|
65
|
+
f"Expected type '{expected_type}', "
|
|
66
|
+
f"got {types}"
|
|
67
|
+
),
|
|
68
|
+
details={"artifacts": artifacts},
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
# 检查内容
|
|
72
|
+
if content_regex:
|
|
73
|
+
contents = [a.get("content", "") for a in artifacts]
|
|
74
|
+
content_match = any(re.search(content_regex, c) for c in contents)
|
|
75
|
+
if not content_match:
|
|
76
|
+
return GraderResult(
|
|
77
|
+
grader_name=self.name,
|
|
78
|
+
grader_type=GraderType.ARTIFACT,
|
|
79
|
+
score=0.3,
|
|
80
|
+
passed=threshold <= 0.3,
|
|
81
|
+
explanation=f"Content does not match pattern: {content_regex}",
|
|
82
|
+
details={"artifacts": artifacts},
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
return GraderResult(
|
|
86
|
+
grader_name=self.name,
|
|
87
|
+
grader_type=GraderType.ARTIFACT,
|
|
88
|
+
score=1.0,
|
|
89
|
+
passed=True,
|
|
90
|
+
explanation=f"Artifact check passed: {len(artifacts)} artifact(s)",
|
|
91
|
+
details={"artifacts": artifacts},
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
def _extract_artifacts(
|
|
95
|
+
self,
|
|
96
|
+
trial: TrialResult,
|
|
97
|
+
spans: list[dict[str, Any]],
|
|
98
|
+
) -> list[dict[str, Any]]:
|
|
99
|
+
"""从 outcome 或 spans 中提取产物"""
|
|
100
|
+
# 优先从 outcome 获取
|
|
101
|
+
artifacts = trial.outcome.get("artifacts", [])
|
|
102
|
+
if artifacts:
|
|
103
|
+
return artifacts
|
|
104
|
+
|
|
105
|
+
# 从 spans 提取
|
|
106
|
+
return [
|
|
107
|
+
{
|
|
108
|
+
"type": span.get("attributes", {}).get("agenthub.artifact_type", ""),
|
|
109
|
+
"id": span.get("attributes", {}).get("agenthub.artifact_id", ""),
|
|
110
|
+
"content": span.get("attributes", {}).get("agenthub.content", ""),
|
|
111
|
+
}
|
|
112
|
+
for span in spans
|
|
113
|
+
if "artifact.create" in span.get("name", "")
|
|
114
|
+
]
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Code-based grader — deterministic scoring via string/regex matching.
|
|
3
|
+
|
|
4
|
+
Supports:
|
|
5
|
+
- contains: substring match
|
|
6
|
+
- not_contains: substring absence
|
|
7
|
+
- regex: regular expression match
|
|
8
|
+
- exact: exact string equality
|
|
9
|
+
|
|
10
|
+
Config schema:
|
|
11
|
+
{
|
|
12
|
+
"checks": [
|
|
13
|
+
{"type": "contains", "value": "def hello", "target": "transcript"},
|
|
14
|
+
{"type": "regex", "value": "class \\w+:", "target": "outcome"},
|
|
15
|
+
],
|
|
16
|
+
"threshold": 1.0 # fraction of checks that must pass
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
Target can be: "transcript" | "outcome" | "spans"
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import json
|
|
25
|
+
import re
|
|
26
|
+
from typing import Any
|
|
27
|
+
|
|
28
|
+
from agent_eval.core.contract import EvalContext
|
|
29
|
+
from agent_eval.core.types import EvalTask, GraderResult, GraderType, TrialResult
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class CodeBasedGrader:
|
|
33
|
+
"""通用确定性评分器"""
|
|
34
|
+
|
|
35
|
+
name = "code_based"
|
|
36
|
+
|
|
37
|
+
async def grade(
|
|
38
|
+
self,
|
|
39
|
+
trial: TrialResult,
|
|
40
|
+
spans: list[dict[str, Any]],
|
|
41
|
+
task: EvalTask,
|
|
42
|
+
context: EvalContext | None = None,
|
|
43
|
+
) -> GraderResult:
|
|
44
|
+
config = task.get_grader_config(self.name)
|
|
45
|
+
checks = config.get("checks", [])
|
|
46
|
+
threshold = config.get("threshold", 1.0)
|
|
47
|
+
|
|
48
|
+
if not checks:
|
|
49
|
+
return GraderResult(
|
|
50
|
+
grader_name=self.name,
|
|
51
|
+
grader_type=GraderType.CODE,
|
|
52
|
+
score=1.0,
|
|
53
|
+
passed=True,
|
|
54
|
+
explanation="No checks configured, auto-pass",
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
passed_count = 0
|
|
58
|
+
details: list[dict[str, Any]] = []
|
|
59
|
+
|
|
60
|
+
for check in checks:
|
|
61
|
+
check_type = check.get("type", "contains")
|
|
62
|
+
target = check.get("target", "transcript")
|
|
63
|
+
value = check.get("value", "")
|
|
64
|
+
|
|
65
|
+
# 获取目标文本
|
|
66
|
+
if target == "transcript":
|
|
67
|
+
text = json.dumps(trial.transcript, ensure_ascii=False)
|
|
68
|
+
elif target == "outcome":
|
|
69
|
+
text = json.dumps(trial.outcome, ensure_ascii=False)
|
|
70
|
+
elif target == "spans":
|
|
71
|
+
text = json.dumps(spans, ensure_ascii=False)
|
|
72
|
+
else:
|
|
73
|
+
text = ""
|
|
74
|
+
|
|
75
|
+
# 执行检查
|
|
76
|
+
if check_type == "contains":
|
|
77
|
+
ok = value in text
|
|
78
|
+
elif check_type == "not_contains":
|
|
79
|
+
ok = value not in text
|
|
80
|
+
elif check_type == "regex":
|
|
81
|
+
ok = bool(re.search(value, text))
|
|
82
|
+
elif check_type == "exact":
|
|
83
|
+
ok = value == text
|
|
84
|
+
else:
|
|
85
|
+
ok = False
|
|
86
|
+
|
|
87
|
+
if ok:
|
|
88
|
+
passed_count += 1
|
|
89
|
+
details.append({"check": check, "passed": ok})
|
|
90
|
+
|
|
91
|
+
total = len(checks)
|
|
92
|
+
score = passed_count / total if total > 0 else 1.0
|
|
93
|
+
|
|
94
|
+
return GraderResult(
|
|
95
|
+
grader_name=self.name,
|
|
96
|
+
grader_type=GraderType.CODE,
|
|
97
|
+
score=score,
|
|
98
|
+
passed=score >= threshold,
|
|
99
|
+
explanation=f"{passed_count}/{total} checks passed",
|
|
100
|
+
details={"checks": details},
|
|
101
|
+
)
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Human grader — routes scoring to human experts with pending semantics.
|
|
3
|
+
|
|
4
|
+
Semantics (design decision D5): ``grade()`` returns IMMEDIATELY with a pending
|
|
5
|
+
result (score=0, passed=False, ``details.status="pending"``, confidence=0) and
|
|
6
|
+
persists a score request to Storage. The run completes normally; pending
|
|
7
|
+
trials are listed separately in the summary and excluded from pass rates.
|
|
8
|
+
Scores come back later via ``POST /api/eval/runs/{run_id}/human-scores``.
|
|
9
|
+
|
|
10
|
+
Config schema:
|
|
11
|
+
{
|
|
12
|
+
"threshold": 0.7, # optional, defaults to task.score_threshold
|
|
13
|
+
"instructions": "...", # optional guidance shown to the reviewer
|
|
14
|
+
}
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import time
|
|
20
|
+
from typing import Any
|
|
21
|
+
|
|
22
|
+
from agent_eval.core.contract import EvalContext
|
|
23
|
+
from agent_eval.core.types import EvalTask, GraderResult, GraderType, TrialResult
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class HumanGrader:
|
|
27
|
+
"""
|
|
28
|
+
人工评分器 — pending 语义, 不阻塞 run 完成。
|
|
29
|
+
|
|
30
|
+
可选注入 Storage (EvalRunner 构造时自动注入): 评分请求通过
|
|
31
|
+
``save_human_score_request`` 落库, 供 Dashboard (change ②) 拉取。
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
name = "human"
|
|
35
|
+
|
|
36
|
+
def __init__(self, storage: Any | None = None):
|
|
37
|
+
self.storage = storage
|
|
38
|
+
|
|
39
|
+
async def grade(
|
|
40
|
+
self,
|
|
41
|
+
trial: TrialResult,
|
|
42
|
+
spans: list[dict[str, Any]],
|
|
43
|
+
task: EvalTask,
|
|
44
|
+
context: EvalContext | None = None,
|
|
45
|
+
) -> GraderResult:
|
|
46
|
+
config = task.get_grader_config(self.name)
|
|
47
|
+
|
|
48
|
+
request: dict[str, Any] = {
|
|
49
|
+
"run_id": context.run_id if context else "",
|
|
50
|
+
"task_id": task.id,
|
|
51
|
+
"trial_index": trial.trial_index,
|
|
52
|
+
"grader_name": self.name,
|
|
53
|
+
"prompt": task.prompt,
|
|
54
|
+
"instructions": config.get("instructions", ""),
|
|
55
|
+
"transcript": trial.transcript,
|
|
56
|
+
"outcome": trial.outcome,
|
|
57
|
+
"created_at": time.time() * 1000,
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
# 评分请求写入 Storage (自定义 Storage 未实现该可选方法时跳过)
|
|
61
|
+
if self.storage is not None:
|
|
62
|
+
save = getattr(self.storage, "save_human_score_request", None)
|
|
63
|
+
if save is not None:
|
|
64
|
+
await save(request)
|
|
65
|
+
|
|
66
|
+
return GraderResult(
|
|
67
|
+
grader_name=self.name,
|
|
68
|
+
grader_type=GraderType.CUSTOM,
|
|
69
|
+
score=0.0,
|
|
70
|
+
passed=False,
|
|
71
|
+
explanation="等待人工评分",
|
|
72
|
+
details={
|
|
73
|
+
"status": "pending",
|
|
74
|
+
"request": request,
|
|
75
|
+
},
|
|
76
|
+
confidence=0.0,
|
|
77
|
+
)
|