aeval-framework 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. aeval_framework-0.1.0.dist-info/METADATA +42 -0
  2. aeval_framework-0.1.0.dist-info/RECORD +63 -0
  3. aeval_framework-0.1.0.dist-info/WHEEL +4 -0
  4. aeval_framework-0.1.0.dist-info/entry_points.txt +2 -0
  5. agent_eval/__init__.py +14 -0
  6. agent_eval/api/__init__.py +14 -0
  7. agent_eval/api/app.py +82 -0
  8. agent_eval/api/events.py +96 -0
  9. agent_eval/api/routes/__init__.py +0 -0
  10. agent_eval/api/routes/datasets.py +441 -0
  11. agent_eval/api/routes/graders.py +19 -0
  12. agent_eval/api/routes/metrics.py +49 -0
  13. agent_eval/api/routes/runs.py +573 -0
  14. agent_eval/api/routes/suites.py +84 -0
  15. agent_eval/api/routes/tasks.py +114 -0
  16. agent_eval/api/standalone.py +105 -0
  17. agent_eval/cli.py +455 -0
  18. agent_eval/core/__init__.py +48 -0
  19. agent_eval/core/contract.py +296 -0
  20. agent_eval/core/metrics.py +184 -0
  21. agent_eval/core/runner.py +868 -0
  22. agent_eval/core/suite.py +60 -0
  23. agent_eval/core/types.py +227 -0
  24. agent_eval/dataset/__init__.py +31 -0
  25. agent_eval/dataset/models.py +199 -0
  26. agent_eval/dataset/quality.py +194 -0
  27. agent_eval/dataset/sources/__init__.py +45 -0
  28. agent_eval/dataset/sources/llm_generator.py +219 -0
  29. agent_eval/dataset/sources/manual.py +172 -0
  30. agent_eval/dataset/sources/regression.py +201 -0
  31. agent_eval/dataset/sources/trace_mining.py +277 -0
  32. agent_eval/dataset/storage.py +342 -0
  33. agent_eval/dataset/version.py +72 -0
  34. agent_eval/examples/__init__.py +0 -0
  35. agent_eval/examples/basic_usage.py +175 -0
  36. agent_eval/examples/mock_runner.py +195 -0
  37. agent_eval/graders/__init__.py +91 -0
  38. agent_eval/graders/artifact_check.py +114 -0
  39. agent_eval/graders/code_based.py +101 -0
  40. agent_eval/graders/human.py +77 -0
  41. agent_eval/graders/metric.py +142 -0
  42. agent_eval/graders/model_based.py +179 -0
  43. agent_eval/graders/state_check.py +106 -0
  44. agent_eval/graders/step_level.py +116 -0
  45. agent_eval/graders/tool_calls.py +102 -0
  46. agent_eval/graders/transcript.py +86 -0
  47. agent_eval/metrics/__init__.py +110 -0
  48. agent_eval/metrics/answer_relevancy.py +57 -0
  49. agent_eval/metrics/base.py +155 -0
  50. agent_eval/metrics/batch_evaluation.py +267 -0
  51. agent_eval/metrics/context_precision.py +62 -0
  52. agent_eval/metrics/context_recall.py +71 -0
  53. agent_eval/metrics/faithfulness.py +72 -0
  54. agent_eval/metrics/llm_judge.py +100 -0
  55. agent_eval/metrics/prompt_metric.py +150 -0
  56. agent_eval/metrics/pytest_plugin.py +308 -0
  57. agent_eval/metrics/report.py +149 -0
  58. agent_eval/metrics/synthetic_data.py +203 -0
  59. agent_eval/storage/__init__.py +17 -0
  60. agent_eval/storage/memory.py +95 -0
  61. agent_eval/storage/sqlite.py +240 -0
  62. agent_eval/trace/__init__.py +16 -0
  63. agent_eval/trace/phoenix.py +144 -0
@@ -0,0 +1,195 @@
1
+ """
2
+ Mock AgentRunner for testing and demonstration.
3
+
4
+ Simulates an agent by returning predefined results, with per-task scripted
5
+ behaviors for exercising the framework's failure paths.
6
+
7
+ Usage:
8
+ from agent_eval.examples.mock_runner import MockAgentRunner
9
+
10
+ # Random behavior (demo)
11
+ runner = EvalRunner(agent_runner=MockAgentRunner())
12
+
13
+ # Scripted behavior (tests): each task consumes its behavior list in
14
+ # order across calls; the last entry repeats once exhausted.
15
+ agent = MockAgentRunner(
16
+ latency_range=(0.0, 0.01),
17
+ script={
18
+ "task_ok": ["success"],
19
+ "task_flaky": ["transient", "transient", "success"],
20
+ "task_dead": ["failure"],
21
+ "task_slow": ["timeout"],
22
+ },
23
+ )
24
+ """
25
+
26
+ from __future__ import annotations
27
+
28
+ import asyncio
29
+ import random
30
+ import uuid
31
+ from typing import Any
32
+
33
+ from agent_eval.core.contract import TransientError
34
+ from agent_eval.core.types import EvalTask
35
+
36
+
37
+ class MockAgentRunner:
38
+ """
39
+ 模拟 AgentRunner。
40
+
41
+ 用于测试和演示框架功能,无需真实 Agent 系统。
42
+ 支持脚本化场景: success / failure / transient / timeout。
43
+ """
44
+
45
+ def __init__(
46
+ self,
47
+ success_rate: float = 0.7,
48
+ latency_range: tuple[float, float] = (0.1, 0.5),
49
+ script: dict[str, list[str]] | None = None,
50
+ timeout_duration: float = 10.0,
51
+ ):
52
+ """
53
+ Args:
54
+ success_rate: 随机模式下的模拟成功率 (0.0-1.0)
55
+ latency_range: 模拟延迟范围 (秒)
56
+ script: task_id → 行为序列 ("success"|"failure"|"transient"|"timeout"),
57
+ 逐次调用消耗, 耗尽后重复最后一项
58
+ timeout_duration: "timeout" 行为的挂起时长 (秒),
59
+ 配合 EvalRunner(per_trial_timeout=...) 触发超时
60
+ """
61
+ self.success_rate = success_rate
62
+ self.latency_range = latency_range
63
+ self.script = script or {}
64
+ self.timeout_duration = timeout_duration
65
+ self.call_counts: dict[str, int] = {}
66
+
67
+ def _next_behavior(self, task_id: str, call_index: int) -> str | None:
68
+ """取该 task 指定调用的脚本行为 (无脚本返回 None = 随机模式)"""
69
+ behaviors = self.script.get(task_id)
70
+ if not behaviors:
71
+ return None
72
+ return behaviors[min(call_index, len(behaviors) - 1)]
73
+
74
+ async def run(
75
+ self,
76
+ task: EvalTask,
77
+ ) -> tuple[str, list[dict[str, Any]], dict[str, Any]]:
78
+ """
79
+ 模拟 Agent 执行。
80
+
81
+ Returns:
82
+ trace_id, transcript, outcome
83
+ """
84
+ index = self.call_counts.get(task.id, 0)
85
+ self.call_counts[task.id] = index + 1
86
+ behavior = self._next_behavior(task.id, index)
87
+
88
+ # 超时场景: 长时间挂起, 由框架的 per_trial_timeout 打断
89
+ if behavior == "timeout":
90
+ await asyncio.sleep(self.timeout_duration)
91
+
92
+ # 模拟延迟
93
+ latency = random.uniform(*self.latency_range)
94
+ await asyncio.sleep(latency)
95
+
96
+ # 瞬态错误场景: 框架按指数退避重试
97
+ if behavior == "transient":
98
+ raise TransientError(
99
+ f"Mock transient failure for {task.id} (call {index + 1})"
100
+ )
101
+
102
+ # 生成 trace_id (编码 task id, 便于 MockTraceProvider 关联 spans)
103
+ trace_id = f"trace_{task.id}_{uuid.uuid4().hex[:8]}"
104
+
105
+ # 构建 transcript
106
+ transcript = [
107
+ {
108
+ "role": "user",
109
+ "content": task.prompt,
110
+ },
111
+ {
112
+ "role": "assistant",
113
+ "content": f"Mock response for task: {task.id}",
114
+ },
115
+ ]
116
+
117
+ # 构建 outcome (模拟成功/失败)
118
+ if behavior == "success":
119
+ success = True
120
+ elif behavior == "failure":
121
+ success = False
122
+ else:
123
+ success = random.random() < self.success_rate
124
+
125
+ outcome: dict[str, Any] = {
126
+ "success": success,
127
+ "files": {
128
+ "output.py": f"# Mock output for {task.id}\ndef hello(): pass\n",
129
+ },
130
+ "artifacts": [
131
+ {
132
+ "type": "code_file",
133
+ "id": f"art_{uuid.uuid4().hex[:8]}",
134
+ "content": f"# Generated code for {task.id}",
135
+ }
136
+ ] if success else [],
137
+ }
138
+
139
+ return trace_id, transcript, outcome
140
+
141
+
142
+ class MockTraceProvider:
143
+ """模拟 TraceProvider
144
+
145
+ 可选按 task id 关联 span 数据: trace_id 形如 "trace_{task_id}_{suffix}"
146
+ 时返回 spans_by_task[task_id] (若已配置), 否则返回默认 spans。
147
+ """
148
+
149
+ def __init__(
150
+ self,
151
+ spans_by_task: dict[str, list[dict[str, Any]]] | None = None,
152
+ default_spans: list[dict[str, Any]] | None = None,
153
+ ):
154
+ self.spans_by_task = spans_by_task or {}
155
+ self.default_spans = default_spans or self._build_default_spans()
156
+
157
+ @staticmethod
158
+ def _build_default_spans() -> list[dict[str, Any]]:
159
+ return [
160
+ {
161
+ "name": "agent.turn",
162
+ "attributes": {
163
+ "agenthub.total_tokens": 150,
164
+ },
165
+ "start_time": "2026-08-29T10:00:00Z",
166
+ "end_time": "2026-08-29T10:00:01Z",
167
+ "status": {"status_code": "OK"},
168
+ },
169
+ {
170
+ "name": "tool.call",
171
+ "attributes": {
172
+ "agenthub.tool_name": "fs_write",
173
+ "agenthub.success": True,
174
+ },
175
+ "start_time": "2026-08-29T10:00:01Z",
176
+ "end_time": "2026-08-29T10:00:02Z",
177
+ "status": {"status_code": "OK"},
178
+ },
179
+ ]
180
+
181
+ async def get_spans(self, trace_id: str) -> list[dict[str, Any]]:
182
+ """返回模拟 span 数据 (优先按 task id 匹配)"""
183
+ if trace_id.startswith("trace_"):
184
+ remainder = trace_id[len("trace_"):]
185
+ for task_id, spans in self.spans_by_task.items():
186
+ if remainder == task_id or remainder.startswith(f"{task_id}_"):
187
+ return spans
188
+ return self.default_spans
189
+
190
+ async def get_trace_ids(
191
+ self,
192
+ filters: dict[str, Any] | None = None,
193
+ limit: int = 100,
194
+ ) -> list[str]:
195
+ return [f"trace_mock_{i}" for i in range(min(limit, 5))]
@@ -0,0 +1,91 @@
1
+ """
2
+ Built-in graders for the Aeval evaluation framework.
3
+
4
+ Provides 9 built-in graders covering agent eval scenarios:
5
+ - code_based: Deterministic checks (string/regex matching)
6
+ - model_based: LLM-as-Judge
7
+ - state_check: Environment state verification
8
+ - tool_calls: Tool call validation
9
+ - transcript: Transcript analysis (turns/tokens/redundancy)
10
+ - artifact_check: Artifact verification
11
+ - human: Human expert scoring (pending semantics, async score submission)
12
+ - step_level: Step-level evaluation (expected_trace comparison)
13
+ - metric: LLM 输出质量指标分发 (按 config.metric_name 路由到 metrics 注册表)
14
+
15
+ Usage:
16
+ from agent_eval.graders import DEFAULT_GRADERS, get_grader_catalog
17
+
18
+ runner = EvalRunner(
19
+ agent_runner=my_runner,
20
+ graders=DEFAULT_GRADERS, # Use all built-in graders
21
+ )
22
+
23
+ # API listing (name/type/description)
24
+ catalog = get_grader_catalog()
25
+ """
26
+
27
+ from typing import Any
28
+
29
+ from agent_eval.core.types import GraderType
30
+ from agent_eval.graders.artifact_check import ArtifactCheckGrader
31
+ from agent_eval.graders.code_based import CodeBasedGrader
32
+ from agent_eval.graders.human import HumanGrader
33
+ from agent_eval.graders.metric import MetricGrader
34
+ from agent_eval.graders.model_based import ModelBasedGrader
35
+ from agent_eval.graders.state_check import StateCheckGrader
36
+ from agent_eval.graders.step_level import StepLevelGrader
37
+ from agent_eval.graders.tool_calls import ToolCallsGrader
38
+ from agent_eval.graders.transcript import TranscriptGrader
39
+
40
+ # 注册表: name → {grader 实例, 类型, 描述} (供 API 列举与默认装配)
41
+ GRADER_REGISTRY: dict[str, dict[str, Any]] = {}
42
+
43
+
44
+ def _register(grader: Any, grader_type: GraderType, description: str) -> None:
45
+ GRADER_REGISTRY[grader.name] = {
46
+ "grader": grader,
47
+ "type": grader_type,
48
+ "description": description,
49
+ }
50
+
51
+
52
+ _register(CodeBasedGrader(), GraderType.CODE, "确定性评分:字符串/正则/精确匹配检查")
53
+ _register(ModelBasedGrader(), GraderType.MODEL, "LLM-as-Judge 评分")
54
+ _register(StateCheckGrader(), GraderType.STATE, "环境状态检查")
55
+ _register(ToolCallsGrader(), GraderType.TOOL_CALLS, "工具调用验证 (必须/禁止调用)")
56
+ _register(TranscriptGrader(), GraderType.TRANSCRIPT, "转录记录分析 (轮次/Token 冗余)")
57
+ _register(ArtifactCheckGrader(), GraderType.ARTIFACT, "产物检查 (类型/内容正则)")
58
+ _register(HumanGrader(), GraderType.CUSTOM, "人工评分 (pending 语义, 异步回传)")
59
+ _register(StepLevelGrader(), GraderType.CUSTOM, "步骤级评估 (expected_trace 对照)")
60
+ _register(MetricGrader(), GraderType.METRIC, "LLM 输出质量指标 (按 metric_name 从注册表分发)")
61
+
62
+ # 默认内置 grader 实例列表
63
+ DEFAULT_GRADERS = [entry["grader"] for entry in GRADER_REGISTRY.values()]
64
+
65
+
66
+ def get_grader_catalog() -> list[dict[str, str]]:
67
+ """列出可用 grader (name/type/description), 供 GET /graders 使用"""
68
+ return [
69
+ {
70
+ "name": name,
71
+ "type": entry["type"].value,
72
+ "description": entry["description"],
73
+ }
74
+ for name, entry in GRADER_REGISTRY.items()
75
+ ]
76
+
77
+
78
+ __all__ = [
79
+ "DEFAULT_GRADERS",
80
+ "GRADER_REGISTRY",
81
+ "get_grader_catalog",
82
+ "CodeBasedGrader",
83
+ "ModelBasedGrader",
84
+ "StateCheckGrader",
85
+ "ToolCallsGrader",
86
+ "TranscriptGrader",
87
+ "ArtifactCheckGrader",
88
+ "HumanGrader",
89
+ "StepLevelGrader",
90
+ "MetricGrader",
91
+ ]
@@ -0,0 +1,114 @@
1
+ """
2
+ Artifact-check grader — artifact verification.
3
+
4
+ Validates artifacts produced by the agent:
5
+ - Artifact exists
6
+ - Artifact type matches expected
7
+ - Artifact content matches regex pattern
8
+
9
+ Config schema:
10
+ {
11
+ "expected_type": "code_file",
12
+ "content_regex": "def \\w+\\(",
13
+ "threshold": 1.0
14
+ }
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import re
20
+ from typing import Any
21
+
22
+ from agent_eval.core.contract import EvalContext
23
+ from agent_eval.core.types import EvalTask, GraderResult, GraderType, TrialResult
24
+
25
+
26
+ class ArtifactCheckGrader:
27
+ """产物检查评分器"""
28
+
29
+ name = "artifact_check"
30
+
31
+ async def grade(
32
+ self,
33
+ trial: TrialResult,
34
+ spans: list[dict[str, Any]],
35
+ task: EvalTask,
36
+ context: EvalContext | None = None,
37
+ ) -> GraderResult:
38
+ config = task.get_grader_config(self.name)
39
+ expected_type = config.get("expected_type")
40
+ content_regex = config.get("content_regex")
41
+ threshold = config.get("threshold", 1.0)
42
+
43
+ # 从 outcome 或 spans 提取产物
44
+ artifacts = self._extract_artifacts(trial, spans)
45
+
46
+ if not artifacts:
47
+ return GraderResult(
48
+ grader_name=self.name,
49
+ grader_type=GraderType.ARTIFACT,
50
+ score=0.0,
51
+ passed=False,
52
+ explanation="No artifacts produced",
53
+ )
54
+
55
+ # 检查类型
56
+ if expected_type:
57
+ types = [a.get("type", "") for a in artifacts]
58
+ if expected_type not in types:
59
+ return GraderResult(
60
+ grader_name=self.name,
61
+ grader_type=GraderType.ARTIFACT,
62
+ score=0.0,
63
+ passed=False,
64
+ explanation=(
65
+ f"Expected type '{expected_type}', "
66
+ f"got {types}"
67
+ ),
68
+ details={"artifacts": artifacts},
69
+ )
70
+
71
+ # 检查内容
72
+ if content_regex:
73
+ contents = [a.get("content", "") for a in artifacts]
74
+ content_match = any(re.search(content_regex, c) for c in contents)
75
+ if not content_match:
76
+ return GraderResult(
77
+ grader_name=self.name,
78
+ grader_type=GraderType.ARTIFACT,
79
+ score=0.3,
80
+ passed=threshold <= 0.3,
81
+ explanation=f"Content does not match pattern: {content_regex}",
82
+ details={"artifacts": artifacts},
83
+ )
84
+
85
+ return GraderResult(
86
+ grader_name=self.name,
87
+ grader_type=GraderType.ARTIFACT,
88
+ score=1.0,
89
+ passed=True,
90
+ explanation=f"Artifact check passed: {len(artifacts)} artifact(s)",
91
+ details={"artifacts": artifacts},
92
+ )
93
+
94
+ def _extract_artifacts(
95
+ self,
96
+ trial: TrialResult,
97
+ spans: list[dict[str, Any]],
98
+ ) -> list[dict[str, Any]]:
99
+ """从 outcome 或 spans 中提取产物"""
100
+ # 优先从 outcome 获取
101
+ artifacts = trial.outcome.get("artifacts", [])
102
+ if artifacts:
103
+ return artifacts
104
+
105
+ # 从 spans 提取
106
+ return [
107
+ {
108
+ "type": span.get("attributes", {}).get("agenthub.artifact_type", ""),
109
+ "id": span.get("attributes", {}).get("agenthub.artifact_id", ""),
110
+ "content": span.get("attributes", {}).get("agenthub.content", ""),
111
+ }
112
+ for span in spans
113
+ if "artifact.create" in span.get("name", "")
114
+ ]
@@ -0,0 +1,101 @@
1
+ """
2
+ Code-based grader — deterministic scoring via string/regex matching.
3
+
4
+ Supports:
5
+ - contains: substring match
6
+ - not_contains: substring absence
7
+ - regex: regular expression match
8
+ - exact: exact string equality
9
+
10
+ Config schema:
11
+ {
12
+ "checks": [
13
+ {"type": "contains", "value": "def hello", "target": "transcript"},
14
+ {"type": "regex", "value": "class \\w+:", "target": "outcome"},
15
+ ],
16
+ "threshold": 1.0 # fraction of checks that must pass
17
+ }
18
+
19
+ Target can be: "transcript" | "outcome" | "spans"
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import json
25
+ import re
26
+ from typing import Any
27
+
28
+ from agent_eval.core.contract import EvalContext
29
+ from agent_eval.core.types import EvalTask, GraderResult, GraderType, TrialResult
30
+
31
+
32
+ class CodeBasedGrader:
33
+ """通用确定性评分器"""
34
+
35
+ name = "code_based"
36
+
37
+ async def grade(
38
+ self,
39
+ trial: TrialResult,
40
+ spans: list[dict[str, Any]],
41
+ task: EvalTask,
42
+ context: EvalContext | None = None,
43
+ ) -> GraderResult:
44
+ config = task.get_grader_config(self.name)
45
+ checks = config.get("checks", [])
46
+ threshold = config.get("threshold", 1.0)
47
+
48
+ if not checks:
49
+ return GraderResult(
50
+ grader_name=self.name,
51
+ grader_type=GraderType.CODE,
52
+ score=1.0,
53
+ passed=True,
54
+ explanation="No checks configured, auto-pass",
55
+ )
56
+
57
+ passed_count = 0
58
+ details: list[dict[str, Any]] = []
59
+
60
+ for check in checks:
61
+ check_type = check.get("type", "contains")
62
+ target = check.get("target", "transcript")
63
+ value = check.get("value", "")
64
+
65
+ # 获取目标文本
66
+ if target == "transcript":
67
+ text = json.dumps(trial.transcript, ensure_ascii=False)
68
+ elif target == "outcome":
69
+ text = json.dumps(trial.outcome, ensure_ascii=False)
70
+ elif target == "spans":
71
+ text = json.dumps(spans, ensure_ascii=False)
72
+ else:
73
+ text = ""
74
+
75
+ # 执行检查
76
+ if check_type == "contains":
77
+ ok = value in text
78
+ elif check_type == "not_contains":
79
+ ok = value not in text
80
+ elif check_type == "regex":
81
+ ok = bool(re.search(value, text))
82
+ elif check_type == "exact":
83
+ ok = value == text
84
+ else:
85
+ ok = False
86
+
87
+ if ok:
88
+ passed_count += 1
89
+ details.append({"check": check, "passed": ok})
90
+
91
+ total = len(checks)
92
+ score = passed_count / total if total > 0 else 1.0
93
+
94
+ return GraderResult(
95
+ grader_name=self.name,
96
+ grader_type=GraderType.CODE,
97
+ score=score,
98
+ passed=score >= threshold,
99
+ explanation=f"{passed_count}/{total} checks passed",
100
+ details={"checks": details},
101
+ )
@@ -0,0 +1,77 @@
1
+ """
2
+ Human grader — routes scoring to human experts with pending semantics.
3
+
4
+ Semantics (design decision D5): ``grade()`` returns IMMEDIATELY with a pending
5
+ result (score=0, passed=False, ``details.status="pending"``, confidence=0) and
6
+ persists a score request to Storage. The run completes normally; pending
7
+ trials are listed separately in the summary and excluded from pass rates.
8
+ Scores come back later via ``POST /api/eval/runs/{run_id}/human-scores``.
9
+
10
+ Config schema:
11
+ {
12
+ "threshold": 0.7, # optional, defaults to task.score_threshold
13
+ "instructions": "...", # optional guidance shown to the reviewer
14
+ }
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import time
20
+ from typing import Any
21
+
22
+ from agent_eval.core.contract import EvalContext
23
+ from agent_eval.core.types import EvalTask, GraderResult, GraderType, TrialResult
24
+
25
+
26
+ class HumanGrader:
27
+ """
28
+ 人工评分器 — pending 语义, 不阻塞 run 完成。
29
+
30
+ 可选注入 Storage (EvalRunner 构造时自动注入): 评分请求通过
31
+ ``save_human_score_request`` 落库, 供 Dashboard (change ②) 拉取。
32
+ """
33
+
34
+ name = "human"
35
+
36
+ def __init__(self, storage: Any | None = None):
37
+ self.storage = storage
38
+
39
+ async def grade(
40
+ self,
41
+ trial: TrialResult,
42
+ spans: list[dict[str, Any]],
43
+ task: EvalTask,
44
+ context: EvalContext | None = None,
45
+ ) -> GraderResult:
46
+ config = task.get_grader_config(self.name)
47
+
48
+ request: dict[str, Any] = {
49
+ "run_id": context.run_id if context else "",
50
+ "task_id": task.id,
51
+ "trial_index": trial.trial_index,
52
+ "grader_name": self.name,
53
+ "prompt": task.prompt,
54
+ "instructions": config.get("instructions", ""),
55
+ "transcript": trial.transcript,
56
+ "outcome": trial.outcome,
57
+ "created_at": time.time() * 1000,
58
+ }
59
+
60
+ # 评分请求写入 Storage (自定义 Storage 未实现该可选方法时跳过)
61
+ if self.storage is not None:
62
+ save = getattr(self.storage, "save_human_score_request", None)
63
+ if save is not None:
64
+ await save(request)
65
+
66
+ return GraderResult(
67
+ grader_name=self.name,
68
+ grader_type=GraderType.CUSTOM,
69
+ score=0.0,
70
+ passed=False,
71
+ explanation="等待人工评分",
72
+ details={
73
+ "status": "pending",
74
+ "request": request,
75
+ },
76
+ confidence=0.0,
77
+ )