ai-eval-scope 0.1.5__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (231) hide show
  1. agent_eval/__init__.py +3 -0
  2. agent_eval/agent/__init__.py +23 -0
  3. agent_eval/agent/callbacks.py +149 -0
  4. agent_eval/agent/eval_tools.py +27 -0
  5. agent_eval/agent/evaluation_agent.py +43 -0
  6. agent_eval/agent/execution_agent.py +409 -0
  7. agent_eval/agent/hooks.py +294 -0
  8. agent_eval/agent/model_bridge.py +95 -0
  9. agent_eval/agent/package_agent.py +381 -0
  10. agent_eval/agent/package_tools.py +364 -0
  11. agent_eval/agent/plan_parser.py +80 -0
  12. agent_eval/agent/protocol_tools.py +204 -0
  13. agent_eval/agent/session.py +235 -0
  14. agent_eval/agent/sut_tools.py +392 -0
  15. agent_eval/agent/tools.py +88 -0
  16. agent_eval/assets/configs/execution_agent_prompts.yaml +61 -0
  17. agent_eval/assets/configs/package_agent_prompts.yaml +69 -0
  18. agent_eval/assets/configs/summary_prompt.yaml +47 -0
  19. agent_eval/assets/configs/sut_config.example.yaml +34 -0
  20. agent_eval/assets/datasets/dataset_index.yaml +182 -0
  21. agent_eval/assets/packages/chat/1.0.0/agent_eval.yaml +16 -0
  22. agent_eval/assets/packages/chat/1.0.0/metrics/policy.yaml +89 -0
  23. agent_eval/assets/packages/chat/1.0.0/prompts/chat_answer_consistency.yaml +133 -0
  24. agent_eval/assets/packages/chat/1.0.0/prompts/chat_answer_extract.yaml +53 -0
  25. agent_eval/assets/packages/chat/1.0.0/prompts/chat_answer_quality.yaml +146 -0
  26. agent_eval/assets/packages/chat/1.0.0/rules/chat-quality.yaml +61 -0
  27. agent_eval/assets/packages/chat/1.0.0/sut_configs/sasan-agent.yaml +27 -0
  28. agent_eval/assets/packages/chat/1.0.0/task_sets/default.yaml +194 -0
  29. agent_eval/assets/packages/code/1.0.0/agent_eval.yaml +14 -0
  30. agent_eval/assets/packages/code/1.0.0/metrics/policy.yaml +69 -0
  31. agent_eval/assets/packages/code/1.0.0/prompts/code_correctness.yaml +94 -0
  32. agent_eval/assets/packages/code/1.0.0/prompts/code_style.yaml +92 -0
  33. agent_eval/assets/packages/code/1.0.0/rules/code-quality.yaml +51 -0
  34. agent_eval/assets/packages/courseware/1.0.0/agent_eval.yaml +12 -0
  35. agent_eval/assets/packages/courseware/1.0.0/datasets/_defaults.yaml +90 -0
  36. agent_eval/assets/packages/courseware/1.0.0/datasets/biology.yaml +2174 -0
  37. agent_eval/assets/packages/courseware/1.0.0/datasets/chemistry.yaml +2946 -0
  38. agent_eval/assets/packages/courseware/1.0.0/datasets/chinese.yaml +6934 -0
  39. agent_eval/assets/packages/courseware/1.0.0/datasets/geography.yaml +2045 -0
  40. agent_eval/assets/packages/courseware/1.0.0/datasets/history.yaml +6780 -0
  41. agent_eval/assets/packages/courseware/1.0.0/datasets/math.yaml +672 -0
  42. agent_eval/assets/packages/courseware/1.0.0/datasets/morality.yaml +4171 -0
  43. agent_eval/assets/packages/courseware/1.0.0/datasets/physics.yaml +4224 -0
  44. agent_eval/assets/packages/courseware/1.0.0/metrics/policy.yaml +112 -0
  45. agent_eval/assets/packages/courseware/1.0.0/prompts/chronological_order.yaml +123 -0
  46. agent_eval/assets/packages/courseware/1.0.0/prompts/content_diversity.yaml +190 -0
  47. agent_eval/assets/packages/courseware/1.0.0/prompts/content_verdict.yaml +82 -0
  48. agent_eval/assets/packages/courseware/1.0.0/prompts/depth_preference.yaml +187 -0
  49. agent_eval/assets/packages/courseware/1.0.0/prompts/fact_verdict.yaml +82 -0
  50. agent_eval/assets/packages/courseware/1.0.0/prompts/info_accuracy.yaml +97 -0
  51. agent_eval/assets/packages/courseware/1.0.0/prompts/knowledge_extract.yaml +45 -0
  52. agent_eval/assets/packages/courseware/1.0.0/prompts/knowledge_extract_constants.yaml +47 -0
  53. agent_eval/assets/packages/courseware/1.0.0/prompts/logical_consistency.yaml +145 -0
  54. agent_eval/assets/packages/courseware/1.0.0/prompts/pedagogical_logic.yaml +189 -0
  55. agent_eval/assets/packages/courseware/1.0.0/prompts/request_fulfillment.yaml +188 -0
  56. agent_eval/assets/packages/courseware/1.0.0/prompts/style_preference.yaml +185 -0
  57. agent_eval/assets/packages/courseware/1.0.0/prompts/visual_quality.yaml +86 -0
  58. agent_eval/assets/packages/courseware/1.0.0/rules/coursework-gate.yaml +73 -0
  59. agent_eval/assets/packages/courseware/1.0.0/rules/coursework-quality.yaml +142 -0
  60. agent_eval/assets/packages/courseware/1.0.0/rules/coursework-vision.yaml +157 -0
  61. agent_eval/assets/schemas/dataset_index_schema.json +92 -0
  62. agent_eval/assets/schemas/dataset_schema.json +100 -0
  63. agent_eval/assets/schemas/rule_set_schema.json +462 -0
  64. agent_eval/assets/schemas/task_set_schema.json +52 -0
  65. agent_eval/cli/__init__.py +40 -0
  66. agent_eval/cli/_common.py +230 -0
  67. agent_eval/cli/_env_file.py +40 -0
  68. agent_eval/cli/_stages.py +343 -0
  69. agent_eval/cli/cmds/__init__.py +4 -0
  70. agent_eval/cli/cmds/auth.py +298 -0
  71. agent_eval/cli/cmds/dataset.py +81 -0
  72. agent_eval/cli/cmds/doctor.py +165 -0
  73. agent_eval/cli/cmds/evaluate.py +165 -0
  74. agent_eval/cli/cmds/execute.py +397 -0
  75. agent_eval/cli/cmds/knowledge.py +205 -0
  76. agent_eval/cli/cmds/models.py +177 -0
  77. agent_eval/cli/cmds/open_url.py +73 -0
  78. agent_eval/cli/cmds/pack.py +144 -0
  79. agent_eval/cli/cmds/rule_set.py +62 -0
  80. agent_eval/cli/cmds/runs.py +239 -0
  81. agent_eval/cli/cmds/scenario.py +422 -0
  82. agent_eval/cli/cmds/scenario_agent.py +416 -0
  83. agent_eval/cli/cmds/secrets.py +191 -0
  84. agent_eval/cli/cmds/suite.py +150 -0
  85. agent_eval/cli/cmds/upload.py +252 -0
  86. agent_eval/cli/console/__init__.py +0 -0
  87. agent_eval/cli/console/equiv.py +44 -0
  88. agent_eval/cli/console/output.py +100 -0
  89. agent_eval/cli/console/prompts.py +151 -0
  90. agent_eval/cli/console/render.py +94 -0
  91. agent_eval/cli/main.py +110 -0
  92. agent_eval/cli/workbench/__init__.py +0 -0
  93. agent_eval/cli/workbench/domains/__init__.py +0 -0
  94. agent_eval/cli/workbench/domains/account.py +45 -0
  95. agent_eval/cli/workbench/domains/exec.py +89 -0
  96. agent_eval/cli/workbench/domains/runs.py +29 -0
  97. agent_eval/cli/workbench/domains/scn.py +63 -0
  98. agent_eval/cli/workbench/session.py +118 -0
  99. agent_eval/config/__init__.py +86 -0
  100. agent_eval/config/evaluation.py +131 -0
  101. agent_eval/config/execution.py +56 -0
  102. agent_eval/config/llm.py +154 -0
  103. agent_eval/config/llm_file.py +106 -0
  104. agent_eval/config/llm_resolution.py +171 -0
  105. agent_eval/config/loader.py +220 -0
  106. agent_eval/config/observability.py +44 -0
  107. agent_eval/config/paths.py +125 -0
  108. agent_eval/config/platform_file.py +100 -0
  109. agent_eval/config/reporting.py +25 -0
  110. agent_eval/core/__init__.py +1 -0
  111. agent_eval/core/exceptions.py +245 -0
  112. agent_eval/core/logging.py +81 -0
  113. agent_eval/core/types.py +93 -0
  114. agent_eval/datasets/__init__.py +20 -0
  115. agent_eval/datasets/downloader.py +127 -0
  116. agent_eval/datasets/manager.py +156 -0
  117. agent_eval/datasets/registry.py +78 -0
  118. agent_eval/evaluation/__init__.py +1 -0
  119. agent_eval/evaluation/base.py +100 -0
  120. agent_eval/evaluation/capability.py +66 -0
  121. agent_eval/evaluation/engine.py +489 -0
  122. agent_eval/evaluation/evaluators/__init__.py +56 -0
  123. agent_eval/evaluation/evaluators/commonsense_evaluators.py +1323 -0
  124. agent_eval/evaluation/evaluators/content_completeness.py +252 -0
  125. agent_eval/evaluation/evaluators/format_evaluators.py +335 -0
  126. agent_eval/evaluation/evaluators/formula_normalizer.py +98 -0
  127. agent_eval/evaluation/evaluators/plugins/__init__.py +91 -0
  128. agent_eval/evaluation/evaluators/quality_evaluators.py +646 -0
  129. agent_eval/evaluation/evaluators/scenario/__init__.py +5 -0
  130. agent_eval/evaluation/evaluators/scenario/chat.py +241 -0
  131. agent_eval/evaluation/evaluators/scenario/code.py +48 -0
  132. agent_eval/evaluation/evaluators/vision_evaluators.py +286 -0
  133. agent_eval/evaluation/models.py +207 -0
  134. agent_eval/evaluation/registry.py +99 -0
  135. agent_eval/evaluation/scenario/__init__.py +40 -0
  136. agent_eval/evaluation/scenario/aggregator.py +79 -0
  137. agent_eval/evaluation/scenario/defaults.py +86 -0
  138. agent_eval/evaluation/scenario/expr.py +199 -0
  139. agent_eval/evaluation/scenario/metrics.py +107 -0
  140. agent_eval/evaluation/scenario/models.py +104 -0
  141. agent_eval/evaluation/stage.py +93 -0
  142. agent_eval/evaluation/summary.py +172 -0
  143. agent_eval/evaluation/text_utils.py +331 -0
  144. agent_eval/evaluation/vision/__init__.py +13 -0
  145. agent_eval/evaluation/vision/renderer.py +202 -0
  146. agent_eval/execution/__init__.py +1 -0
  147. agent_eval/execution/auth/__init__.py +7 -0
  148. agent_eval/execution/auth/credentials.py +130 -0
  149. agent_eval/execution/auth/provider.py +186 -0
  150. agent_eval/execution/auth/secrets_store.py +57 -0
  151. agent_eval/execution/auth/session.py +139 -0
  152. agent_eval/execution/channels/__init__.py +5 -0
  153. agent_eval/execution/channels/agent_protocol.py +300 -0
  154. agent_eval/execution/channels/base.py +135 -0
  155. agent_eval/execution/channels/thread_commands.py +290 -0
  156. agent_eval/execution/models.py +192 -0
  157. agent_eval/execution/registry.py +246 -0
  158. agent_eval/execution/task_builder.py +94 -0
  159. agent_eval/execution/utils.py +38 -0
  160. agent_eval/knowledge/__init__.py +73 -0
  161. agent_eval/knowledge/auditor.py +350 -0
  162. agent_eval/knowledge/base.py +56 -0
  163. agent_eval/knowledge/converters/__init__.py +1 -0
  164. agent_eval/knowledge/converters/math_reference.py +40 -0
  165. agent_eval/knowledge/converters/periodic_table.py +148 -0
  166. agent_eval/knowledge/converters/physics_reference.py +41 -0
  167. agent_eval/knowledge/exceptions.py +49 -0
  168. agent_eval/knowledge/extractors/__init__.py +1 -0
  169. agent_eval/knowledge/extractors/llm_extractor.py +152 -0
  170. agent_eval/knowledge/extractors/parsers.py +25 -0
  171. agent_eval/knowledge/manager.py +112 -0
  172. agent_eval/knowledge/merger.py +115 -0
  173. agent_eval/knowledge/models.py +109 -0
  174. agent_eval/knowledge/pipeline.py +128 -0
  175. agent_eval/knowledge/registry.py +119 -0
  176. agent_eval/knowledge/sources/__init__.py +1 -0
  177. agent_eval/knowledge/sources/eval_sources.py +121 -0
  178. agent_eval/knowledge/sources/math_reference.py +193 -0
  179. agent_eval/knowledge/sources/physics_reference.py +151 -0
  180. agent_eval/knowledge/sources/table_sources.py +54 -0
  181. agent_eval/llm/__init__.py +26 -0
  182. agent_eval/llm/client.py +74 -0
  183. agent_eval/llm/factory.py +69 -0
  184. agent_eval/llm/judge/__init__.py +1 -0
  185. agent_eval/llm/judge/file_prompt_store.py +70 -0
  186. agent_eval/llm/judge/orchestrator.py +332 -0
  187. agent_eval/llm/judge/prompt_store.py +69 -0
  188. agent_eval/llm/judge/recorder.py +48 -0
  189. agent_eval/llm/judge/stability.py +93 -0
  190. agent_eval/llm/judge/structured_output.py +110 -0
  191. agent_eval/llm/judge/template_manager.py +191 -0
  192. agent_eval/llm/models.py +164 -0
  193. agent_eval/llm/pool.py +58 -0
  194. agent_eval/llm/providers/__init__.py +1 -0
  195. agent_eval/llm/providers/anthropic.py +229 -0
  196. agent_eval/llm/providers/openai_compat.py +230 -0
  197. agent_eval/llm/tracing.py +144 -0
  198. agent_eval/observability/__init__.py +12 -0
  199. agent_eval/observability/client.py +170 -0
  200. agent_eval/observability/config.py +111 -0
  201. agent_eval/observability/events.py +222 -0
  202. agent_eval/observability/queue.py +163 -0
  203. agent_eval/observability/schemas/ingest.event.v1.json +159 -0
  204. agent_eval/observability/sink.py +386 -0
  205. agent_eval/observability/snapshot.py +87 -0
  206. agent_eval/orchestrator/__init__.py +5 -0
  207. agent_eval/orchestrator/orchestrator.py +697 -0
  208. agent_eval/packages/__init__.py +42 -0
  209. agent_eval/packages/assets.py +169 -0
  210. agent_eval/packages/manager.py +105 -0
  211. agent_eval/packages/manifest.py +122 -0
  212. agent_eval/packages/remote_client.py +127 -0
  213. agent_eval/packages/store.py +183 -0
  214. agent_eval/reporting/__init__.py +5 -0
  215. agent_eval/reporting/report_generator.py +402 -0
  216. agent_eval/rules/__init__.py +1 -0
  217. agent_eval/rules/models.py +272 -0
  218. agent_eval/rules/template.py +79 -0
  219. agent_eval/rules/validation.py +83 -0
  220. agent_eval/sdk/__init__.py +7 -0
  221. agent_eval/sdk/rules.py +120 -0
  222. agent_eval/storage/__init__.py +1 -0
  223. agent_eval/storage/builder.py +317 -0
  224. agent_eval/storage/collector.py +204 -0
  225. agent_eval/storage/package.py +352 -0
  226. agent_eval/storage/workspace.py +186 -0
  227. ai_eval_scope-0.1.5.dist-info/METADATA +121 -0
  228. ai_eval_scope-0.1.5.dist-info/RECORD +231 -0
  229. ai_eval_scope-0.1.5.dist-info/WHEEL +4 -0
  230. ai_eval_scope-0.1.5.dist-info/entry_points.txt +2 -0
  231. ai_eval_scope-0.1.5.dist-info/licenses/LICENSE +21 -0
agent_eval/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """Agent 能力评估系统"""
2
+
3
+ __version__ = "0.1.5"
@@ -0,0 +1,23 @@
1
+ """Agent 模块 — ExecutionAgent(DeepAgents 底座)+ SUT Tools + 回调/会话/模型桥接。
2
+
3
+ heavyweight 依赖(deepagents/langchain)均为 [agent] optional extra 惰性导入,
4
+ 本包导入本身零额外依赖。
5
+ """
6
+
7
+ from agent_eval.agent.callbacks import BudgetGuard, SessionLogCallback
8
+ from agent_eval.agent.hooks import AgentExecutionLog, BudgetController, SessionLogger
9
+ from agent_eval.agent.protocol_tools import AgentProtocolToolServer
10
+ from agent_eval.agent.session import AgentSession, WorkspaceCheckpointer
11
+ from agent_eval.agent.sut_tools import SUTToolServer
12
+
13
+ __all__ = [
14
+ "AgentExecutionLog",
15
+ "AgentProtocolToolServer",
16
+ "AgentSession",
17
+ "BudgetController",
18
+ "BudgetGuard",
19
+ "SessionLogCallback",
20
+ "SessionLogger",
21
+ "SUTToolServer",
22
+ "WorkspaceCheckpointer",
23
+ ]
@@ -0,0 +1,149 @@
1
+ """LangGraph 回调 — 预算护栏与会话日志注入(arch/03 §7a.6 v4.6)。
2
+
3
+ BudgetGuard / SessionLogCallback 实现 LangChain 回调协议的
4
+ on_llm_end / on_tool_start / on_tool_end 方法,经
5
+ ainvoke(config={"callbacks": [...]}) 注入 DeepAgents 图执行。
6
+ [agent] extra 环境下继承 BaseCallbackHandler(真 langchain 回调管理器会
7
+ 访问 run_inline / ignore_* 等基类属性,裸鸭子类型会 AttributeError);
8
+ 纯 mock 测试环境退化为 object。
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from typing import Any
14
+
15
+ from agent_eval.agent.hooks import BudgetController, SessionLogger
16
+ from agent_eval.agent.sut_tools import HTTP_RAW_MAX_CHARS
17
+ from agent_eval.core.exceptions import BudgetExceededError
18
+
19
+ try:
20
+ from langchain_core.callbacks import BaseCallbackHandler as _LCBaseCallbackHandler
21
+ except ImportError: # pragma: no cover — langchain 属 [agent] extra,可选
22
+ _LCBaseCallbackHandler = object # type: ignore[assignment,misc]
23
+
24
+
25
+ def _extract_usage(response: Any) -> dict[str, int] | None:
26
+ """从 LLMResult 兼容形态提取 token 用量。
27
+
28
+ 依次尝试(LangChain 新旧版本 / dict 兼容):
29
+ 1. generations[i][j].message.usage_metadata(ChatModel 新式)
30
+ 2. response["usage_metadata"](dict 形态)
31
+ 3. response.llm_output["token_usage"](旧式 LLMResult)
32
+ """
33
+ for generation in getattr(response, "generations", []) or []:
34
+ if isinstance(generation, list):
35
+ for chunk in generation:
36
+ message = getattr(chunk, "message", None)
37
+ usage = getattr(message, "usage_metadata", None)
38
+ if isinstance(usage, dict):
39
+ return usage
40
+ if isinstance(response, dict):
41
+ usage = response.get("usage_metadata")
42
+ if isinstance(usage, dict):
43
+ return usage
44
+ llm_output = response.get("llm_output") or {}
45
+ if isinstance(llm_output, dict):
46
+ return llm_output.get("token_usage")
47
+ llm_output = getattr(response, "llm_output", None)
48
+ if isinstance(llm_output, dict):
49
+ return llm_output.get("token_usage")
50
+ return None
51
+
52
+
53
+ def _normalize_usage(usage: dict[str, int] | None) -> tuple[int, int]:
54
+ """归一化 token 用量为 (input_tokens, output_tokens)。
55
+
56
+ 新式 usage_metadata: input_tokens/output_tokens;
57
+ 旧式 token_usage: prompt_tokens/completion_tokens。
58
+ """
59
+ if not usage:
60
+ return 0, 0
61
+ input_tokens = usage.get("input_tokens", usage.get("prompt_tokens", 0)) or 0
62
+ output_tokens = usage.get("output_tokens", usage.get("completion_tokens", 0)) or 0
63
+ return int(input_tokens), int(output_tokens)
64
+
65
+
66
+ class BudgetGuard(_LCBaseCallbackHandler): # type: ignore[misc]
67
+ """预算护栏回调:on_llm_end 累计 token/成本,超限抛 BudgetExceededError 终止图执行。
68
+
69
+ 成本估算依赖可选 pricing 配置({"input_per_1k": x, "output_per_1k": y},
70
+ 可来自 provider extra_params.pricing);未配置时仅累计 token,成本记 0。
71
+ """
72
+
73
+ def __init__(
74
+ self,
75
+ max_budget_usd: float,
76
+ controller: BudgetController | None = None,
77
+ *,
78
+ pricing: dict[str, float] | None = None,
79
+ ) -> None:
80
+ self.controller = controller or BudgetController(max_budget_usd)
81
+ self.pricing = pricing or {}
82
+ self.llm_calls = 0
83
+
84
+ def on_llm_end(self, response: Any, *, run_id: Any = None, **kwargs: Any) -> None:
85
+ """LLM 调用结束:计量 token/成本并检查预算。"""
86
+ self.llm_calls += 1
87
+ input_tokens, output_tokens = _normalize_usage(_extract_usage(response))
88
+ cost_usd = 0.0
89
+ if self.pricing:
90
+ cost_usd = input_tokens / 1000 * self.pricing.get(
91
+ "input_per_1k", 0.0
92
+ ) + output_tokens / 1000 * self.pricing.get("output_per_1k", 0.0)
93
+ state = self.controller.record(cost_usd, input_tokens + output_tokens)
94
+ if state == "exceeded":
95
+ raise BudgetExceededError(
96
+ f"Agent 执行超出预算: 已花费 ${self.controller.spent_usd:.4f} "
97
+ f"(上限 ${self.controller.max_budget_usd})",
98
+ )
99
+
100
+ @property
101
+ def spent_usd(self) -> float:
102
+ """当前累计成本。"""
103
+ return self.controller.spent_usd
104
+
105
+ @property
106
+ def total_tokens(self) -> int:
107
+ """当前累计 token。"""
108
+ return self.controller.total_tokens
109
+
110
+
111
+ class SessionLogCallback(_LCBaseCallbackHandler): # type: ignore[misc]
112
+ """会话日志回调:on_tool_start/on_tool_end → SessionLogger 结构化日志。"""
113
+
114
+ def __init__(self, logger: SessionLogger) -> None:
115
+ self.logger = logger
116
+ # run_id → (call_id, tool_name),on_tool_end 无工具名,靠配对还原
117
+ self._active: dict[str, tuple[str, str]] = {}
118
+
119
+ def on_tool_start(
120
+ self,
121
+ serialized: dict[str, Any] | None,
122
+ input_str: str,
123
+ *,
124
+ run_id: Any = None,
125
+ inputs: dict[str, Any] | None = None,
126
+ **kwargs: Any,
127
+ ) -> None:
128
+ """工具调用开始:记录 tool_call 事件。"""
129
+ name = ""
130
+ if isinstance(serialized, dict):
131
+ name = serialized.get("name") or ""
132
+ tool_input = dict(inputs or {})
133
+ if not tool_input and input_str:
134
+ tool_input = {"input": input_str[:HTTP_RAW_MAX_CHARS]}
135
+ call_id = self.logger.log_tool_call(name or "unknown", tool_input)
136
+ self._active[str(run_id)] = (call_id, name or "unknown")
137
+
138
+ def on_tool_end(self, output: Any, *, run_id: Any = None, **kwargs: Any) -> None:
139
+ """工具调用结束:记录 tool_result 事件。"""
140
+ call_id, name = self._active.pop(str(run_id), (None, "unknown"))
141
+ summary = {"output": str(output)[:HTTP_RAW_MAX_CHARS]}
142
+ self.logger.log_tool_result(name, summary, call_id, status="success")
143
+
144
+ def on_tool_error(
145
+ self, error: BaseException | None, *, run_id: Any = None, **kwargs: Any
146
+ ) -> None:
147
+ """工具调用异常:记录 error 状态的 tool_result。"""
148
+ call_id, name = self._active.pop(str(run_id), (None, "unknown"))
149
+ self.logger.log_tool_result(name, {"error": str(error)}, call_id, status="error")
@@ -0,0 +1,27 @@
1
+ """EvalToolServer — 评估器工具(MCP)。
2
+
3
+ 为 EvaluationAgent 提供评估器调用能力的 MCP Tools。
4
+ 后续迭代实现,当前为骨架文件。
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from typing import Any
10
+
11
+
12
+ class EvalToolServer:
13
+ """评估器工具服务器,为 EvaluationAgent 提供评估器调用能力。
14
+
15
+ 注意:本类将在后续迭代中完整实现。当前仅提供接口骨架。
16
+ """
17
+
18
+ def __init__(self, config: dict[str, Any] | None = None) -> None:
19
+ self.config = config or {}
20
+
21
+ def get_tool_names(self) -> list[str]:
22
+ """返回所有注册的评估工具名称。"""
23
+ return [
24
+ "evaluate_constraint",
25
+ "get_evaluator_info",
26
+ "list_evaluators",
27
+ ]
@@ -0,0 +1,43 @@
1
+ """EvaluationAgent — Agent 自适应评估(可选评估模式)。
2
+
3
+ 读取 Markdown 评测计划,自主完成评估任务。
4
+ Sprint 8+ 实现,当前为骨架文件。
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from typing import Any
10
+
11
+
12
+ class EvaluationAgent:
13
+ """EvaluationAgent — Agent 自适应评估。
14
+
15
+ 通过 EvalToolServer(MCP Tools)调用评估器,
16
+ 根据 Markdown 评测计划自适应执行评估流程。
17
+
18
+ 注意:本类将在后续迭代中完整实现。当前仅提供接口骨架。
19
+ """
20
+
21
+ def __init__(self, config: dict[str, Any] | None = None) -> None:
22
+ self.config = config or {}
23
+
24
+ async def evaluate(
25
+ self,
26
+ package_path: str,
27
+ rule_set_path: str,
28
+ eval_plan_path: str | None = None,
29
+ ) -> dict[str, Any]:
30
+ """执行自适应评估。
31
+
32
+ Args:
33
+ package_path: ExecutionPackage 目录路径。
34
+ rule_set_path: 规则集文件路径。
35
+ eval_plan_path: Markdown 评测计划路径(可选)。
36
+
37
+ Returns:
38
+ 评估结果字典。
39
+
40
+ Raises:
41
+ NotImplementedError: 当前迭代尚未实现。
42
+ """
43
+ raise NotImplementedError("EvaluationAgent.evaluate() 将在后续迭代中实现。")
@@ -0,0 +1,409 @@
1
+ """ExecutionAgent — 基于 DeepAgents(Python)的执行 Agent(arch/03 §三 v4.6)。
2
+
3
+ 端到端驱动评测执行流程:理解任务、调用 SUT Tools、处理错误、收集结果、
4
+ 生成 ExecutionPackage。底座为 deepagents 的 create_deep_agent(惰性导入,
5
+ [agent] optional extra);模型经 build_chat_model 从角色注册表双协议桥接;
6
+ BudgetGuard/SessionLogCallback 以 LangGraph 回调注入(预算/结构化日志)。
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import json
12
+ from datetime import UTC, datetime
13
+ from functools import lru_cache
14
+ from pathlib import Path
15
+ from typing import Any
16
+
17
+ import yaml
18
+
19
+ from agent_eval.agent.callbacks import BudgetGuard, SessionLogCallback
20
+ from agent_eval.agent.hooks import SessionLogger
21
+ from agent_eval.agent.model_bridge import build_chat_model
22
+ from agent_eval.agent.session import AgentSession
23
+ from agent_eval.agent.sut_tools import SUTToolServer
24
+ from agent_eval.config.paths import PACKAGE_ROOT
25
+ from agent_eval.core.exceptions import (
26
+ AgentError,
27
+ AgentTimeoutError,
28
+ BudgetExceededError,
29
+ )
30
+ from agent_eval.execution.models import AgentConfig, ProcessMetrics, Task, TaskSet
31
+ from agent_eval.storage.package import ExecutionPackage, generate_run_id
32
+
33
+ # 执行 Agent 提示词资产(prompt 在 YAML 中维护,不 hardcode;对齐 summary_prompt.yaml 惯例)
34
+ _PROMPTS_PATH = PACKAGE_ROOT / "assets" / "configs" / "execution_agent_prompts.yaml"
35
+
36
+
37
+ @lru_cache(maxsize=1)
38
+ def _load_prompts() -> dict[str, Any]:
39
+ """加载 execution_agent_prompts.yaml → {system_prompt, task_prompt}。"""
40
+ try:
41
+ data = yaml.safe_load(_PROMPTS_PATH.read_text(encoding="utf-8"))
42
+ except (OSError, yaml.YAMLError) as e:
43
+ raise AgentError(
44
+ f"执行 Agent 提示词资产损坏: {_PROMPTS_PATH}({e})",
45
+ details={"path": str(_PROMPTS_PATH)},
46
+ ) from e
47
+ if not isinstance(data, dict) or not data.get("system_prompt") or not data.get("task_prompt"):
48
+ raise AgentError(
49
+ f"执行 Agent 提示词资产结构不完整(需 system_prompt/task_prompt 两段): {_PROMPTS_PATH}",
50
+ details={"path": str(_PROMPTS_PATH)},
51
+ )
52
+ return data
53
+
54
+
55
+ def _now_iso() -> str:
56
+ return datetime.now(UTC).isoformat()
57
+
58
+
59
+ def _is_recursion_error(error: BaseException) -> bool:
60
+ """识别 LangGraph GraphRecursionError(按类名,避免硬依赖 langgraph)。"""
61
+ return type(error).__name__ == "GraphRecursionError"
62
+
63
+
64
+ class ExecutionAgent:
65
+ """基于 DeepAgents 的执行 Agent,端到端驱动评测执行流程。
66
+
67
+ - 模型:LLM 角色注册表(arch/06 §4.6)→ build_chat_model() 构造 ChatModel(双协议,模型无关)
68
+ - 工具:SUT Tools 经 LangChain Tool 显式绑定(白名单),未绑定工具不可用
69
+ - 预算:BudgetGuard 回调(on_llm_end 累计 token/成本,超限抛 BudgetExceededError)
70
+ - 状态:单任务单发 ainvoke,不接 checkpointer(见 _build_graph 说明)
71
+ """
72
+
73
+ def __init__(
74
+ self,
75
+ config: AgentConfig,
76
+ sut_tools: SUTToolServer | None = None,
77
+ *,
78
+ extra_tool_servers: list[Any] | None = None,
79
+ ) -> None:
80
+ """初始化 ExecutionAgent(DeepAgents 图惰性装配,导入本类无需 [agent] extra)。
81
+
82
+ Args:
83
+ config: Agent 配置(轮次/预算/llm_role/workspace 等)。
84
+ sut_tools: SUT 工具注册表;缺省按 config.sut_tools_config 构建。
85
+ extra_tool_servers: 追加工具注册表(如 AgentProtocolToolServer,
86
+ arch/03 §4.0.6 语义工具面),与 SUT Tools 一同显式绑定。
87
+ """
88
+ self.config = config
89
+ self.sut_tools = sut_tools or SUTToolServer(
90
+ config.sut_tools_config, workspace_dir=config.workspace_dir
91
+ )
92
+ # 外部注入的注册表(如 AgentProtocolToolServer)同样以 config.workspace_dir
93
+ # 为落盘根——write_package/collect_results 的目的地不交给 LLM 决定
94
+ self.sut_tools.workspace_dir = Path(config.workspace_dir)
95
+ self.tool_servers: list[Any] = [self.sut_tools, *(extra_tool_servers or [])]
96
+ self._graph: Any = None
97
+
98
+ # ─── 对外入口 ───
99
+
100
+ async def run_task_set(
101
+ self, task_set: TaskSet, *, run_id: str | None = None
102
+ ) -> tuple[str, list[ExecutionPackage]]:
103
+ """批量执行任务集(共享 run_id),返回 (run_id, 执行包列表)。
104
+
105
+ run_id 可由调用方(CLI)注入——用于运行清单登记与外部关联。
106
+ """
107
+ run_id = run_id or generate_run_id()
108
+ packages = [await self.run_task(task, run_id=run_id) for task in task_set.tasks]
109
+ return run_id, packages
110
+
111
+ async def run_task(self, task: Task, *, run_id: str | None = None) -> ExecutionPackage:
112
+ """执行单个任务,返回 ExecutionPackage。
113
+
114
+ Raises:
115
+ AgentTimeoutError: 超过轮次限制(已写入含部分结果的执行包)。
116
+ BudgetExceededError: 超过 max_budget_usd(已写入错误执行包)。
117
+ AgentError: DeepAgents/LangGraph 运行时异常(已写入错误执行包)。
118
+ """
119
+ run_id = run_id or generate_run_id()
120
+ workspace = Path(self.config.workspace_dir)
121
+ # W7(arch/03 §7a.8):执行包归位 runs/{run_id}/packages/{task_id}——
122
+ # 挂 run_id 与 agent_logs/results 同层可关联(旧布局 workspace/{task_id}
123
+ # 同名重跑互相覆盖且无法归属运行)。write_package 的目的地由
124
+ # sut_tools.workspace_dir 决定,逐 run 注入包根。
125
+ run_packages_root = workspace / "runs" / run_id / "packages"
126
+ self.sut_tools.workspace_dir = run_packages_root
127
+ package_dir = run_packages_root / task.id
128
+ log_dir = workspace / "runs" / run_id / "agent_logs"
129
+
130
+ logger = SessionLogger(run_id, task.id, log_dir=log_dir)
131
+ guard = BudgetGuard(self.config.max_budget_usd)
132
+ logger.log_start(task.input, task.constraints)
133
+
134
+ try:
135
+ graph = self._ensure_graph()
136
+ result = await graph.ainvoke(
137
+ {"messages": [{"role": "user", "content": self._build_task_prompt(task)}]},
138
+ config={
139
+ "recursion_limit": self.config.max_turns * 2,
140
+ "callbacks": [SessionLogCallback(logger), guard],
141
+ },
142
+ )
143
+ except (BudgetExceededError, AgentTimeoutError, AgentError) as e:
144
+ await self._abort(logger, guard, task, package_dir, e)
145
+ raise
146
+ except Exception as e:
147
+ error: AgentError
148
+ if _is_recursion_error(e):
149
+ error = AgentTimeoutError(
150
+ f"Agent 执行超过轮次限制(max_turns={self.config.max_turns})"
151
+ )
152
+ else:
153
+ error = AgentError(f"Agent 会话异常中断: {e}")
154
+ await self._abort(logger, guard, task, package_dir, error)
155
+ raise error from e
156
+
157
+ session = AgentSession.from_messages(result.get("messages", []))
158
+ session.started_at = logger.started_at or _now_iso()
159
+ session.finished_at = _now_iso()
160
+ package = await self._build_package(session, task, package_dir)
161
+ logger.log_end(
162
+ cost_usd=guard.spent_usd,
163
+ tokens_used=guard.total_tokens,
164
+ turns_used=session.turns_used,
165
+ )
166
+ logger.close()
167
+ return package
168
+
169
+ # ─── DeepAgents 图装配 ───
170
+
171
+ def _ensure_graph(self) -> Any:
172
+ """惰性装配 DeepAgents 图(首次 run_task 时构建并复用)。"""
173
+ if self._graph is None:
174
+ self._graph = self._build_graph()
175
+ return self._graph
176
+
177
+ def _build_graph(self) -> Any:
178
+ """构建 DeepAgents 图:模型桥接 + 工具显式绑定。
179
+
180
+ 不接 checkpointer:单任务单发 ainvoke 无恢复需求;且 langgraph 会经
181
+ put/put_writes/get_tuple 高频触达 saver,文件全量读写实现会拖垮执行
182
+ (v4.6.3 实测 CPU 空转)。WorkspaceCheckpointer 保留为独立组件,
183
+ 待 B4 实现语义正确的真 saver(增量写 + pending_writes 按
184
+ checkpoint_id 索引)后再接回。
185
+ """
186
+ try:
187
+ from deepagents import create_deep_agent
188
+ except ImportError:
189
+ raise AgentError(
190
+ "ExecutionAgent 需要 deepagents(DeepAgents 底座,见 arch/03 §3.2)。"
191
+ "请执行: pip install 'agent-eval[agent]'",
192
+ details={"missing_module": "deepagents"},
193
+ ) from None
194
+ tools = [tool for server in self.tool_servers for tool in server.to_langchain_tools()]
195
+ return create_deep_agent(
196
+ model=build_chat_model(self.config.llm_role, self.config.model),
197
+ tools=tools,
198
+ system_prompt=self._build_system_prompt(),
199
+ )
200
+
201
+ # ─── Prompt 构建(arch/03 §3.3/§3.4) ───
202
+
203
+ def _describe_all_tools(self) -> str:
204
+ """汇总全部工具注册表(SUT Tools + 追加注册表)的描述清单。"""
205
+ return "\n".join(server.describe_tools() for server in self.tool_servers)
206
+
207
+ def _build_system_prompt(self) -> str:
208
+ """System Prompt:角色职责 + 可用工具 + 执行规则 + 输出规范(模板见 execution_agent_prompts.yaml)。"""
209
+ template: str = _load_prompts()["system_prompt"]
210
+ return template.format(
211
+ tools=self._describe_all_tools(),
212
+ max_turns=self.config.max_turns,
213
+ max_retries=self.config.max_retries,
214
+ )
215
+
216
+ def _build_task_prompt(self, task: Task) -> str:
217
+ """Task Prompt:任务输入/转发指令/预期/约束 + 目录模式 + 写包指令(模板见 execution_agent_prompts.yaml)。
218
+
219
+ forward 段提供确定性的纯文本转发内容——执行 Agent 不再依赖 LLM
220
+ 自行从 JSON 结构中提取 instruction(此前行为不一致,有时传整个 dict)。
221
+ """
222
+ segments: dict[str, str] = _load_prompts()["task_prompt"]
223
+ package_dir = Path(self.config.workspace_dir) / task.id
224
+ instruction_text = self._extract_instruction(task)
225
+ parts = [
226
+ segments["header"].format(task_id=task.id),
227
+ segments["input"].format(
228
+ task_input=json.dumps(task.input, ensure_ascii=False, indent=2)
229
+ ),
230
+ segments["forward"].format(instruction_text=instruction_text),
231
+ ]
232
+ if task.expected:
233
+ parts.append(
234
+ segments["expected"].format(
235
+ expected=json.dumps(task.expected, ensure_ascii=False, indent=2)
236
+ )
237
+ )
238
+ if task.constraints:
239
+ parts.append(
240
+ segments["constraints"].format(
241
+ constraints=json.dumps(task.constraints, ensure_ascii=False, indent=2)
242
+ )
243
+ )
244
+ if task.input_mode == "directory" and task.directory_path:
245
+ parts.append(
246
+ segments["directory_mode"].format(
247
+ directory_path=task.directory_path,
248
+ file_patterns=task.file_patterns,
249
+ )
250
+ )
251
+ parts.append(segments["footer"].format(package_dir=package_dir))
252
+ return "\n\n".join(p.rstrip("\n") for p in parts)
253
+
254
+ @staticmethod
255
+ def _extract_instruction(task: Task) -> str:
256
+ """从 task.input 提取纯文本指令(agent_run 的确定转发内容)。
257
+
258
+ - dict 型 input:取 instruction 字段(缺失时取第一个字符串值)
259
+ - str 型 input:直接返回
260
+ """
261
+ if isinstance(task.input, dict):
262
+ text = task.input.get("instruction", "")
263
+ if not text:
264
+ # 兼容无 instruction 键的 input:取第一个非空字符串值
265
+ for v in task.input.values():
266
+ if isinstance(v, str) and v.strip():
267
+ text = v
268
+ break
269
+ return str(text).strip()
270
+ return str(task.input).strip()
271
+
272
+ # ─── ExecutionPackage 构建 ───
273
+
274
+ async def _build_package(
275
+ self,
276
+ session: AgentSession,
277
+ task: Task,
278
+ package_dir: Path,
279
+ ) -> ExecutionPackage:
280
+ """从 Agent 会话构建 ExecutionPackage。
281
+
282
+ Agent 已调用 write_package 时直接加载其产物;否则写入兜底失败包
283
+ (status=failed,error=未写包说明),再补齐 task/trace/metrics。
284
+ """
285
+ if not (package_dir / "manifest.json").exists():
286
+ await self.sut_tools.write_package(
287
+ task_id=task.id,
288
+ success=False,
289
+ error="Agent 未调用 write_package,已由 ExecutionAgent 兜底写包",
290
+ )
291
+ self._ensure_task_file(task, package_dir)
292
+ self._ensure_trace_file(session, package_dir)
293
+ self._ensure_answer_file(package_dir)
294
+ self._ensure_metrics_file(session, package_dir)
295
+ return ExecutionPackage.load(package_dir)
296
+
297
+ async def _abort(
298
+ self,
299
+ logger: SessionLogger,
300
+ guard: BudgetGuard,
301
+ task: Task,
302
+ package_dir: Path,
303
+ error: BaseException,
304
+ ) -> None:
305
+ """异常路径统一收尾:日志 + 错误执行包(保留 Agent 已写的部分结果)。"""
306
+ logger.log_error(type(error).__name__, str(error))
307
+ if not (package_dir / "manifest.json").exists():
308
+ await self.sut_tools.write_package(
309
+ task_id=task.id,
310
+ success=False,
311
+ error=str(error),
312
+ )
313
+ logger.log_end(cost_usd=guard.spent_usd, tokens_used=guard.total_tokens)
314
+ logger.close()
315
+
316
+ def _ensure_task_file(self, task: Task, package_dir: Path) -> None:
317
+ task_file = package_dir / "task.json"
318
+ if not task_file.exists():
319
+ task_file.write_text(task.model_dump_json(indent=2), encoding="utf-8")
320
+
321
+ def _ensure_trace_file(self, session: AgentSession, package_dir: Path) -> None:
322
+ """写/补全 trace.json(merge 语义)。
323
+
324
+ LLM 经 write_package 工具已写入 SUT-run 形态(run_id/thread_id/sut_response/
325
+ turns_used…)时保留其字段,仅以 setdefault 补充 Agent 过程统计
326
+ (messages/tool_calls/turns/duration_ms,Sprint 9 v6.0 过程指标数据源);
327
+ 未写时创建完整骨架并回填 SUT 最终回答(v4.6.4)。
328
+ """
329
+ trace_file = package_dir / "trace.json"
330
+ trace: dict[str, Any] = {}
331
+ if trace_file.exists():
332
+ try:
333
+ loaded = json.loads(trace_file.read_text(encoding="utf-8"))
334
+ if isinstance(loaded, dict):
335
+ trace = loaded
336
+ except (OSError, ValueError):
337
+ trace = {}
338
+ duration_ms = 0.0
339
+ try:
340
+ start = datetime.fromisoformat(session.started_at)
341
+ end = datetime.fromisoformat(session.finished_at or _now_iso())
342
+ duration_ms = (end - start).total_seconds() * 1000
343
+ except ValueError:
344
+ pass
345
+ response = trace.setdefault("response", {})
346
+ if not isinstance(response, dict):
347
+ response = trace["response"] = {}
348
+ response.setdefault("messages", len(session.messages))
349
+ response.setdefault("tool_calls", session.tool_call_count)
350
+ response.setdefault("turns", session.turns_used)
351
+ response.setdefault("duration_ms", duration_ms)
352
+ # 回填 SUT 最终回答(trace 只存计数时下游 eval 拿不到评估对象,v4.6.4)
353
+ if "sut" not in response:
354
+ sut_run = self._last_sut_run()
355
+ if sut_run is not None:
356
+ response["sut"] = sut_run
357
+ trace.setdefault(
358
+ "request", {"executor": "ExecutionAgent", "llm_role": self.config.llm_role}
359
+ )
360
+ trace.setdefault("started_at", session.started_at)
361
+ trace.setdefault("finished_at", session.finished_at or _now_iso())
362
+ trace.setdefault("error", None)
363
+ trace_file.write_text(json.dumps(trace, ensure_ascii=False, indent=2), encoding="utf-8")
364
+
365
+ def _last_sut_run(self) -> dict[str, Any] | None:
366
+ """取工具注册表记录的最近一次 SUT run 摘要(AgentProtocolToolServer.last_run)。"""
367
+ for server in self.tool_servers:
368
+ last: dict[str, Any] | None = getattr(server, "last_run", None)
369
+ if last:
370
+ return last
371
+ return None
372
+
373
+ def _ensure_answer_file(self, package_dir: Path) -> None:
374
+ """SUT 回答物化为 output/answer.md(对话型任务无产物文件;评估器按文件收集文本)。"""
375
+ text = (self._last_sut_run() or {}).get("text") or ""
376
+ if not text.strip():
377
+ return
378
+ output_dir = package_dir / "output"
379
+ if output_dir.exists() and any(output_dir.iterdir()):
380
+ return # SUT 已有产物文件,不重复物化
381
+ output_dir.mkdir(parents=True, exist_ok=True)
382
+ (output_dir / "answer.md").write_text(text, encoding="utf-8")
383
+
384
+ def _ensure_metrics_file(self, session: AgentSession, package_dir: Path) -> None:
385
+ """写/补全 metrics.json(merge 语义:保留 LLM 已写字段,setdefault 补过程统计)。"""
386
+ metrics_file = package_dir / "metrics.json"
387
+ metrics: dict[str, Any] = {}
388
+ if metrics_file.exists():
389
+ try:
390
+ loaded = json.loads(metrics_file.read_text(encoding="utf-8"))
391
+ if isinstance(loaded, dict):
392
+ metrics = loaded
393
+ except (OSError, ValueError):
394
+ metrics = {}
395
+ duration_ms = 0.0
396
+ try:
397
+ start = datetime.fromisoformat(session.started_at)
398
+ end = datetime.fromisoformat(session.finished_at or _now_iso())
399
+ duration_ms = (end - start).total_seconds() * 1000
400
+ except ValueError:
401
+ pass
402
+ agent_metrics = ProcessMetrics(
403
+ total_duration_ms=duration_ms,
404
+ steps=len(session.messages),
405
+ tool_calls=session.tool_call_count,
406
+ ).model_dump()
407
+ for k, v in agent_metrics.items():
408
+ metrics.setdefault(k, v)
409
+ metrics_file.write_text(json.dumps(metrics, ensure_ascii=False, indent=2), encoding="utf-8")