aeval-framework 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. aeval_framework-0.1.0.dist-info/METADATA +42 -0
  2. aeval_framework-0.1.0.dist-info/RECORD +63 -0
  3. aeval_framework-0.1.0.dist-info/WHEEL +4 -0
  4. aeval_framework-0.1.0.dist-info/entry_points.txt +2 -0
  5. agent_eval/__init__.py +14 -0
  6. agent_eval/api/__init__.py +14 -0
  7. agent_eval/api/app.py +82 -0
  8. agent_eval/api/events.py +96 -0
  9. agent_eval/api/routes/__init__.py +0 -0
  10. agent_eval/api/routes/datasets.py +441 -0
  11. agent_eval/api/routes/graders.py +19 -0
  12. agent_eval/api/routes/metrics.py +49 -0
  13. agent_eval/api/routes/runs.py +573 -0
  14. agent_eval/api/routes/suites.py +84 -0
  15. agent_eval/api/routes/tasks.py +114 -0
  16. agent_eval/api/standalone.py +105 -0
  17. agent_eval/cli.py +455 -0
  18. agent_eval/core/__init__.py +48 -0
  19. agent_eval/core/contract.py +296 -0
  20. agent_eval/core/metrics.py +184 -0
  21. agent_eval/core/runner.py +868 -0
  22. agent_eval/core/suite.py +60 -0
  23. agent_eval/core/types.py +227 -0
  24. agent_eval/dataset/__init__.py +31 -0
  25. agent_eval/dataset/models.py +199 -0
  26. agent_eval/dataset/quality.py +194 -0
  27. agent_eval/dataset/sources/__init__.py +45 -0
  28. agent_eval/dataset/sources/llm_generator.py +219 -0
  29. agent_eval/dataset/sources/manual.py +172 -0
  30. agent_eval/dataset/sources/regression.py +201 -0
  31. agent_eval/dataset/sources/trace_mining.py +277 -0
  32. agent_eval/dataset/storage.py +342 -0
  33. agent_eval/dataset/version.py +72 -0
  34. agent_eval/examples/__init__.py +0 -0
  35. agent_eval/examples/basic_usage.py +175 -0
  36. agent_eval/examples/mock_runner.py +195 -0
  37. agent_eval/graders/__init__.py +91 -0
  38. agent_eval/graders/artifact_check.py +114 -0
  39. agent_eval/graders/code_based.py +101 -0
  40. agent_eval/graders/human.py +77 -0
  41. agent_eval/graders/metric.py +142 -0
  42. agent_eval/graders/model_based.py +179 -0
  43. agent_eval/graders/state_check.py +106 -0
  44. agent_eval/graders/step_level.py +116 -0
  45. agent_eval/graders/tool_calls.py +102 -0
  46. agent_eval/graders/transcript.py +86 -0
  47. agent_eval/metrics/__init__.py +110 -0
  48. agent_eval/metrics/answer_relevancy.py +57 -0
  49. agent_eval/metrics/base.py +155 -0
  50. agent_eval/metrics/batch_evaluation.py +267 -0
  51. agent_eval/metrics/context_precision.py +62 -0
  52. agent_eval/metrics/context_recall.py +71 -0
  53. agent_eval/metrics/faithfulness.py +72 -0
  54. agent_eval/metrics/llm_judge.py +100 -0
  55. agent_eval/metrics/prompt_metric.py +150 -0
  56. agent_eval/metrics/pytest_plugin.py +308 -0
  57. agent_eval/metrics/report.py +149 -0
  58. agent_eval/metrics/synthetic_data.py +203 -0
  59. agent_eval/storage/__init__.py +17 -0
  60. agent_eval/storage/memory.py +95 -0
  61. agent_eval/storage/sqlite.py +240 -0
  62. agent_eval/trace/__init__.py +16 -0
  63. agent_eval/trace/phoenix.py +144 -0
@@ -0,0 +1,142 @@
1
+ """
2
+ Metric grader — the built-in dispatcher from grader configs to Metric instances.
3
+
4
+ Task grader configs with `type: metric` route through this grader (D1):
5
+ - `name: metric` + `config.metric_name` (single dispatcher per task), or
6
+ - `name: <metric_name>` + `type: metric` (several metrics per task — the
7
+ runner falls back to this dispatcher for unregistered metric-type configs)
8
+
9
+ The metric registry and LLM function are injected by EvalRunner
10
+ (metrics_registry=..., llm_fn=...). Unregistered metric names score 0 with
11
+ an explicit reason; missing LLM configuration surfaces a clear config error
12
+ result instead of crashing the run.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from typing import Any
18
+
19
+ from agent_eval.core.contract import EvalContext
20
+ from agent_eval.core.types import (
21
+ EvalTask,
22
+ GraderConfig,
23
+ GraderResult,
24
+ GraderType,
25
+ TrialResult,
26
+ )
27
+ from agent_eval.metrics.base import Metric, MetricError
28
+ from agent_eval.metrics.llm_judge import LLMFn, LLMJudgeError, LLMNotConfiguredError
29
+
30
+ # 计算失败 (配置/解析/调用) 映射为 0 分结果, 不 crash run (D2)
31
+ _CALC_ERRORS = (LLMNotConfiguredError, LLMJudgeError, MetricError)
32
+
33
+
34
+ class MetricGrader:
35
+ """按 config.metric_name 从注入注册表分发到对应 Metric 计算"""
36
+
37
+ name = "metric"
38
+
39
+ def __init__(
40
+ self,
41
+ metrics_registry: dict[str, Metric] | None = None,
42
+ llm_fn: LLMFn | None = None,
43
+ ):
44
+ # 由 EvalRunner 组合根覆盖注入 (与 storage 注入同一模式)
45
+ self.metrics_registry: dict[str, Metric] = dict(metrics_registry or {})
46
+ self.llm_fn = llm_fn
47
+
48
+ async def grade(
49
+ self,
50
+ trial: TrialResult,
51
+ spans: list[dict[str, Any]],
52
+ task: EvalTask,
53
+ context: EvalContext | None = None,
54
+ ) -> GraderResult:
55
+ config = self._active_config(task, context)
56
+ metric_name = self._metric_name(config)
57
+
58
+ if not metric_name:
59
+ return self._result(config, 0.0, False, "metric grader 缺少 config.metric_name")
60
+
61
+ metric = self.metrics_registry.get(metric_name)
62
+ if metric is None:
63
+ return self._result(
64
+ config, 0.0, False, f"未知指标: {metric_name} (未在 metrics_registry 注册)"
65
+ )
66
+
67
+ kwargs = self._measure_kwargs(config, trial)
68
+ try:
69
+ result = await metric.measure(**kwargs)
70
+ except _CALC_ERRORS as e:
71
+ return self._result(config, 0.0, False, f"配置/计算错误: {e}")
72
+ except Exception as e: # noqa: BLE001 — 指标实现方错误同样不 crash run
73
+ return self._result(config, 0.0, False, f"指标计算异常: {e}")
74
+
75
+ threshold = float(config.config.get("threshold", metric.threshold))
76
+ return self._result(
77
+ config,
78
+ result.score,
79
+ result.score >= threshold,
80
+ result.reason,
81
+ details={
82
+ **result.details,
83
+ "metric": result.name,
84
+ "metric_threshold": result.threshold,
85
+ "grader_threshold": threshold,
86
+ },
87
+ )
88
+
89
+ # ── Config resolution ────────────────────────────────────────────────
90
+
91
+ @staticmethod
92
+ def _active_config(task: EvalTask, context: EvalContext | None) -> GraderConfig:
93
+ """当前生效的 metric 配置: runner 传入优先, 否则取名为 metric 的配置"""
94
+ if context is not None and context.grader_config is not None:
95
+ return context.grader_config
96
+ for g in task.graders:
97
+ if g.type == GraderType.METRIC:
98
+ return g
99
+ raise ValueError(f"task '{task.id}' has no metric grader config")
100
+
101
+ @staticmethod
102
+ def _metric_name(config: GraderConfig) -> str:
103
+ """metric_name 显式配置优先; 命名分发型 (name=指标名) 取配置名"""
104
+ explicit = config.config.get("metric_name")
105
+ if explicit:
106
+ return str(explicit)
107
+ if config.name != MetricGrader.name:
108
+ return config.name
109
+ return ""
110
+
111
+ @staticmethod
112
+ def _measure_kwargs(config: GraderConfig, trial: TrialResult) -> dict[str, Any]:
113
+ """从 trial transcript 与 grader config 提取 measure() 入参"""
114
+ first = trial.transcript[0] if trial.transcript else {}
115
+ last = trial.transcript[-1] if trial.transcript else {}
116
+ prompt = first.get("content", "") if isinstance(first, dict) else ""
117
+ output = last.get("content", "") if isinstance(last, dict) else ""
118
+
119
+ return {
120
+ "input": prompt,
121
+ "actual_output": output,
122
+ "expected_output": config.config.get("expected_output"),
123
+ "context": config.config.get("context"),
124
+ "retrieval_context": config.config.get("retrieval_context"),
125
+ }
126
+
127
+ @staticmethod
128
+ def _result(
129
+ config: GraderConfig,
130
+ score: float,
131
+ passed: bool,
132
+ explanation: str,
133
+ details: dict[str, Any] | None = None,
134
+ ) -> GraderResult:
135
+ return GraderResult(
136
+ grader_name=config.name,
137
+ grader_type=GraderType.METRIC,
138
+ score=max(0.0, min(1.0, score)),
139
+ passed=passed,
140
+ explanation=explanation,
141
+ details=details or {},
142
+ )
@@ -0,0 +1,179 @@
1
+ """
2
+ Model-based grader — LLM-as-Judge scoring.
3
+
4
+ Sends the trial transcript and a rubric to an LLM, which returns
5
+ a score between 0 and 1.
6
+
7
+ Config schema:
8
+ {
9
+ "rubric": "The response must contain...",
10
+ "dimensions": ["correctness", "completeness"],
11
+ "threshold": 0.7,
12
+ "model": "gpt-4o-mini", # optional
13
+ }
14
+
15
+ Requires either:
16
+ - A configured llm_fn callback, or
17
+ - An API key in the environment (OPENAI_API_KEY, etc.)
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import contextlib
23
+ import json
24
+ import os
25
+ from collections.abc import Callable
26
+ from typing import Any
27
+
28
+ from agent_eval.core.contract import EvalContext
29
+ from agent_eval.core.types import EvalTask, GraderResult, GraderType, TrialResult
30
+
31
+ # Type alias for LLM function: (system_prompt, user_message) -> str
32
+ LLMFn = Callable[[str, str], str]
33
+
34
+
35
+ class ModelBasedGrader:
36
+ """LLM-as-Judge 评分器"""
37
+
38
+ name = "model_based"
39
+
40
+ def __init__(self, llm_fn: LLMFn | None = None):
41
+ """
42
+ Args:
43
+ llm_fn: (system_prompt, user_message) → response text.
44
+ If None, uses OpenAI API with OPENAI_API_KEY.
45
+ """
46
+ self._llm_fn = llm_fn
47
+
48
+ async def grade(
49
+ self,
50
+ trial: TrialResult,
51
+ spans: list[dict[str, Any]],
52
+ task: EvalTask,
53
+ context: EvalContext | None = None,
54
+ ) -> GraderResult:
55
+ config = task.get_grader_config(self.name)
56
+ rubric = config.get("rubric", "")
57
+ dimensions = config.get("dimensions", ["quality"])
58
+ threshold = config.get("threshold", 0.7)
59
+
60
+ # Build judge prompt
61
+ prompt = self._build_prompt(trial, rubric, dimensions)
62
+
63
+ # Call LLM
64
+ llm_fn = self._llm_fn or self._default_llm_fn(config)
65
+ try:
66
+ raw = llm_fn("You are an evaluation expert.", prompt)
67
+ except Exception as e:
68
+ return GraderResult(
69
+ grader_name=self.name,
70
+ grader_type=GraderType.MODEL,
71
+ score=0.0,
72
+ passed=False,
73
+ explanation=f"LLM call failed: {e}",
74
+ )
75
+
76
+ # Parse scores
77
+ scores = self._parse_scores(raw, dimensions)
78
+ avg_score = sum(scores.values()) / len(scores) if scores else 0.0
79
+
80
+ return GraderResult(
81
+ grader_name=self.name,
82
+ grader_type=GraderType.MODEL,
83
+ score=avg_score,
84
+ passed=avg_score >= threshold,
85
+ explanation=f"LLM Judge scores: {scores}",
86
+ details={"dimensions": scores, "raw_response": raw},
87
+ )
88
+
89
+ def _build_prompt(
90
+ self,
91
+ trial: TrialResult,
92
+ rubric: str,
93
+ dimensions: list[str],
94
+ ) -> str:
95
+ """构建 judge prompt"""
96
+ input_msg = trial.transcript[0] if trial.transcript else "N/A"
97
+ output_msg = trial.transcript[-1] if trial.transcript else "N/A"
98
+
99
+ # 提取工具调用摘要
100
+ tools_used = list(set(
101
+ msg.get("tool_name", "")
102
+ for msg in trial.transcript
103
+ if msg.get("role") == "tool_call"
104
+ ))
105
+
106
+ dims_json = ", ".join(f'"{d}": 0.0' for d in dimensions)
107
+
108
+ return f"""请根据以下评分标准对 Agent 表现进行评分。
109
+
110
+ ## 评分标准
111
+ {rubric}
112
+
113
+ ## 评分维度
114
+ {", ".join(dimensions)}
115
+
116
+ ## Agent 执行记录
117
+ - 输入: {input_msg}
118
+ - 输出: {output_msg}
119
+ - 使用的工具: {tools_used or "无"}
120
+
121
+ 请以 JSON 格式返回各维度评分 (0.0-1.0):
122
+ ```json
123
+ {{{dims_json}}}
124
+ ```"""
125
+
126
+ def _parse_scores(self, raw: str, dimensions: list[str]) -> dict[str, float]:
127
+ """从 LLM 响应中解析分数"""
128
+ scores: dict[str, float] = {}
129
+
130
+ # 尝试提取 JSON
131
+ try:
132
+ # 查找 JSON 块
133
+ json_start = raw.find("{")
134
+ json_end = raw.rfind("}") + 1
135
+ if json_start >= 0 and json_end > json_start:
136
+ json_str = raw[json_start:json_end]
137
+ parsed = json.loads(json_str)
138
+ for dim in dimensions:
139
+ if dim in parsed:
140
+ with contextlib.suppress(ValueError, TypeError):
141
+ scores[dim] = float(parsed[dim])
142
+ except json.JSONDecodeError:
143
+ pass
144
+
145
+ # 如果解析失败,给所有维度默认分
146
+ if not scores:
147
+ for dim in dimensions:
148
+ scores[dim] = 0.5
149
+
150
+ return scores
151
+
152
+ def _default_llm_fn(self, config: dict[str, Any]) -> LLMFn:
153
+ """创建默认的 LLM 调用函数"""
154
+ model = config.get("model", "gpt-4o-mini")
155
+
156
+ def call_llm(system: str, user: str) -> str:
157
+ try:
158
+ from openai import OpenAI
159
+
160
+ client = OpenAI(api_key=os.getenv("OPENAI_API_KEY"))
161
+ resp = client.chat.completions.create(
162
+ model=model,
163
+ messages=[
164
+ {"role": "system", "content": system},
165
+ {"role": "user", "content": user},
166
+ ],
167
+ temperature=0.0,
168
+ max_tokens=500,
169
+ )
170
+ return resp.choices[0].message.content or ""
171
+ except ImportError as e:
172
+ raise RuntimeError(
173
+ "openai package not installed. "
174
+ "Install with: pip install openai"
175
+ ) from e
176
+ except Exception as e:
177
+ raise RuntimeError(f"OpenAI API call failed: {e}") from e
178
+
179
+ return call_llm
@@ -0,0 +1,106 @@
1
+ """
2
+ State-check grader — environment state verification.
3
+
4
+ Validates the environment state after a trial:
5
+ - file_exists: file is present
6
+ - file_contains: file content contains a substring
7
+ - db_record: database record matches criteria
8
+ - custom: custom check function
9
+
10
+ Config schema:
11
+ {
12
+ "expectations": [
13
+ {"type": "file_exists", "path": "output.py"},
14
+ {"type": "file_contains", "path": "output.py", "value": "def main"},
15
+ {"type": "db_record", "table": "users", "match": {"id": 1}},
16
+ ],
17
+ "threshold": 1.0
18
+ }
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import re
24
+ from typing import Any
25
+
26
+ from agent_eval.core.contract import EvalContext
27
+ from agent_eval.core.types import EvalTask, GraderResult, GraderType, TrialResult
28
+
29
+
30
+ class StateCheckGrader:
31
+ """环境状态检查评分器"""
32
+
33
+ name = "state_check"
34
+
35
+ async def grade(
36
+ self,
37
+ trial: TrialResult,
38
+ spans: list[dict[str, Any]],
39
+ task: EvalTask,
40
+ context: EvalContext | None = None,
41
+ ) -> GraderResult:
42
+ config = task.get_grader_config(self.name)
43
+ expectations = config.get("expectations", [])
44
+ threshold = config.get("threshold", 1.0)
45
+
46
+ if not expectations:
47
+ return GraderResult(
48
+ grader_name=self.name,
49
+ grader_type=GraderType.STATE,
50
+ score=1.0,
51
+ passed=True,
52
+ explanation="No expectations configured, auto-pass",
53
+ )
54
+
55
+ passed_count = 0
56
+ details: list[dict[str, Any]] = []
57
+
58
+ for exp in expectations:
59
+ exp_type = exp.get("type", "file_exists")
60
+ ok = False
61
+
62
+ if exp_type == "file_exists":
63
+ files = trial.outcome.get("files", {})
64
+ ok = exp["path"] in files
65
+
66
+ elif exp_type == "file_contains":
67
+ files = trial.outcome.get("files", {})
68
+ content = files.get(exp["path"], "")
69
+ ok = exp["value"] in content
70
+
71
+ elif exp_type == "file_regex":
72
+ files = trial.outcome.get("files", {})
73
+ content = files.get(exp["path"], "")
74
+ ok = bool(re.search(exp["value"], content))
75
+
76
+ elif exp_type == "db_record":
77
+ records = trial.outcome.get("db_records", [])
78
+ match_criteria = exp.get("match", {})
79
+ ok = any(
80
+ all(r.get(k) == v for k, v in match_criteria.items())
81
+ for r in records
82
+ )
83
+
84
+ elif exp_type == "no_conflict_markers":
85
+ files = trial.outcome.get("files", {})
86
+ content = files.get(exp["path"], "")
87
+ ok = "<<<<<<<" not in content and ">>>>>>>" not in content
88
+
89
+ else:
90
+ ok = False
91
+
92
+ if ok:
93
+ passed_count += 1
94
+ details.append({"expectation": exp, "passed": ok})
95
+
96
+ total = len(expectations)
97
+ score = passed_count / total if total > 0 else 1.0
98
+
99
+ return GraderResult(
100
+ grader_name=self.name,
101
+ grader_type=GraderType.STATE,
102
+ score=score,
103
+ passed=score >= threshold,
104
+ explanation=f"{passed_count}/{total} state checks passed",
105
+ details={"expectations": details},
106
+ )
@@ -0,0 +1,116 @@
1
+ """
2
+ Step-level grader — compares the agent's tool-call sequence against an
3
+ expected trace (design decision D6, first version: exact index-by-index
4
+ comparison only).
5
+
6
+ From the trace spans, extracts the sequence of ``tool.call`` steps (tool name
7
+ from span attributes, falling back to the span name), aligns it with the
8
+ task's ``expected_trace`` config by index, reports the first wrong step and
9
+ scores ``correct_steps / total_steps``.
10
+
11
+ Config schema:
12
+ {
13
+ "expected_trace": ["fs_read", "fs_write", "bash"], # required
14
+ "threshold": 0.7, # optional, pass threshold
15
+ }
16
+
17
+ When ``expected_trace`` is not configured the grader auto-passes (nothing to
18
+ compare against), mirroring code_based's behavior with no checks.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ from typing import Any
24
+
25
+ from agent_eval.core.contract import EvalContext
26
+ from agent_eval.core.types import EvalTask, GraderResult, GraderType, TrialResult
27
+
28
+
29
+ class StepLevelGrader:
30
+ """步骤级评估 — expected_trace 按索引对照, 定位首个错误步骤"""
31
+
32
+ name = "step_level"
33
+
34
+ async def grade(
35
+ self,
36
+ trial: TrialResult,
37
+ spans: list[dict[str, Any]],
38
+ task: EvalTask,
39
+ context: EvalContext | None = None,
40
+ ) -> GraderResult:
41
+ config = task.get_grader_config(self.name)
42
+ expected = config.get("expected_trace")
43
+ threshold = config.get("threshold", 0.7)
44
+
45
+ actual = self._extract_steps(spans)
46
+
47
+ if not expected:
48
+ return GraderResult(
49
+ grader_name=self.name,
50
+ grader_type=GraderType.CUSTOM,
51
+ score=1.0,
52
+ passed=True,
53
+ explanation="No expected_trace configured, auto-pass",
54
+ details={"actual_steps": actual},
55
+ )
56
+
57
+ total = len(expected)
58
+ step_details: list[dict[str, Any]] = []
59
+ first_error: int | None = None
60
+
61
+ for i in range(total):
62
+ expected_step = expected[i]
63
+ actual_step = actual[i] if i < len(actual) else None
64
+ ok = actual_step == expected_step
65
+ if not ok and first_error is None:
66
+ first_error = i
67
+ step_details.append({
68
+ "index": i,
69
+ "expected": expected_step,
70
+ "actual": actual_step,
71
+ "correct": ok,
72
+ })
73
+
74
+ correct_count = sum(1 for s in step_details if s["correct"])
75
+ score = correct_count / total if total > 0 else 1.0
76
+
77
+ explanation = f"{correct_count}/{total} steps correct"
78
+ if first_error is not None:
79
+ explanation += (
80
+ f"; first error at step {first_error}: "
81
+ f"expected '{expected[first_error]}', "
82
+ f"got '{actual[first_error] if first_error < len(actual) else None}'"
83
+ )
84
+
85
+ return GraderResult(
86
+ grader_name=self.name,
87
+ grader_type=GraderType.CUSTOM,
88
+ score=score,
89
+ passed=score >= threshold,
90
+ explanation=explanation,
91
+ details={
92
+ "steps": step_details,
93
+ "first_error_step": first_error,
94
+ "actual_steps": actual,
95
+ "extra_steps": (
96
+ actual[total:] if len(actual) > total else []
97
+ ),
98
+ },
99
+ )
100
+
101
+ @staticmethod
102
+ def _extract_steps(spans: list[dict[str, Any]]) -> list[str]:
103
+ """从 spans 提取 tool.call 步骤序列 (工具名, 回退到 span 名称)"""
104
+ steps: list[str] = []
105
+ for span in spans:
106
+ name = span.get("name", "")
107
+ if "tool.call" not in name and "tool_call" not in name:
108
+ continue
109
+ attrs = span.get("attributes", {}) or {}
110
+ tool_name = (
111
+ attrs.get("agenthub.tool_name")
112
+ or attrs.get("tool_name")
113
+ or name
114
+ )
115
+ steps.append(str(tool_name))
116
+ return steps
@@ -0,0 +1,102 @@
1
+ """
2
+ Tool-calls grader — tool call validation.
3
+
4
+ Validates that the agent:
5
+ - Called required tools
6
+ - Did not use forbidden tools
7
+ - (Optional) Called tools in a specific order
8
+
9
+ Config schema:
10
+ {
11
+ "required_tools": ["fs_read", "fs_write"],
12
+ "forbidden_tools": ["dangerous_tool"],
13
+ "threshold": 1.0
14
+ }
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from typing import Any
20
+
21
+ from agent_eval.core.contract import EvalContext
22
+ from agent_eval.core.types import EvalTask, GraderResult, GraderType, TrialResult
23
+
24
+
25
+ class ToolCallsGrader:
26
+ """工具调用验证评分器"""
27
+
28
+ name = "tool_calls"
29
+
30
+ async def grade(
31
+ self,
32
+ trial: TrialResult,
33
+ spans: list[dict[str, Any]],
34
+ task: EvalTask,
35
+ context: EvalContext | None = None,
36
+ ) -> GraderResult:
37
+ config = task.get_grader_config(self.name)
38
+ required_tools = config.get("required_tools", [])
39
+ forbidden_tools = config.get("forbidden_tools", [])
40
+ threshold = config.get("threshold", 1.0)
41
+
42
+ # 从 spans 提取工具调用
43
+ tool_calls = self._extract_tool_calls(spans)
44
+ used_tools = [tc["name"] for tc in tool_calls]
45
+
46
+ # 检查必须使用的工具
47
+ missing = [t for t in required_tools if t not in used_tools]
48
+ # 检查禁止使用的工具
49
+ violated = [t for t in forbidden_tools if t in used_tools]
50
+
51
+ # 计算分数
52
+ if required_tools:
53
+ found = len(required_tools) - len(missing)
54
+ score = found / len(required_tools)
55
+ else:
56
+ score = 1.0
57
+
58
+ # 违反禁止工具则直接 0 分
59
+ if violated:
60
+ score = 0.0
61
+
62
+ passed = score >= threshold
63
+
64
+ # 构建解释
65
+ parts = []
66
+ if used_tools:
67
+ parts.append(f"Used: {used_tools}")
68
+ if missing:
69
+ parts.append(f"Missing required: {missing}")
70
+ if violated:
71
+ parts.append(f"Violated forbidden: {violated}")
72
+ explanation = "; ".join(parts) if parts else "No tool calls checked"
73
+
74
+ return GraderResult(
75
+ grader_name=self.name,
76
+ grader_type=GraderType.TOOL_CALLS,
77
+ score=score,
78
+ passed=passed,
79
+ explanation=explanation,
80
+ details={
81
+ "tool_calls": tool_calls,
82
+ "used_tools": used_tools,
83
+ "missing": missing,
84
+ "violated": violated,
85
+ },
86
+ )
87
+
88
+ def _extract_tool_calls(self, spans: list[dict[str, Any]]) -> list[dict[str, Any]]:
89
+ """从 spans 中提取工具调用"""
90
+ tool_calls = []
91
+ for span in spans:
92
+ name = span.get("name", "")
93
+ attrs = span.get("attributes", {})
94
+
95
+ if "tool.call" in name or "tool_call" in name:
96
+ tool_calls.append({
97
+ "name": attrs.get("agenthub.tool_name")
98
+ or attrs.get("tool.name", ""),
99
+ "success": attrs.get("agenthub.success", True),
100
+ })
101
+
102
+ return tool_calls