aeval-framework 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. aeval_framework-0.1.0.dist-info/METADATA +42 -0
  2. aeval_framework-0.1.0.dist-info/RECORD +63 -0
  3. aeval_framework-0.1.0.dist-info/WHEEL +4 -0
  4. aeval_framework-0.1.0.dist-info/entry_points.txt +2 -0
  5. agent_eval/__init__.py +14 -0
  6. agent_eval/api/__init__.py +14 -0
  7. agent_eval/api/app.py +82 -0
  8. agent_eval/api/events.py +96 -0
  9. agent_eval/api/routes/__init__.py +0 -0
  10. agent_eval/api/routes/datasets.py +441 -0
  11. agent_eval/api/routes/graders.py +19 -0
  12. agent_eval/api/routes/metrics.py +49 -0
  13. agent_eval/api/routes/runs.py +573 -0
  14. agent_eval/api/routes/suites.py +84 -0
  15. agent_eval/api/routes/tasks.py +114 -0
  16. agent_eval/api/standalone.py +105 -0
  17. agent_eval/cli.py +455 -0
  18. agent_eval/core/__init__.py +48 -0
  19. agent_eval/core/contract.py +296 -0
  20. agent_eval/core/metrics.py +184 -0
  21. agent_eval/core/runner.py +868 -0
  22. agent_eval/core/suite.py +60 -0
  23. agent_eval/core/types.py +227 -0
  24. agent_eval/dataset/__init__.py +31 -0
  25. agent_eval/dataset/models.py +199 -0
  26. agent_eval/dataset/quality.py +194 -0
  27. agent_eval/dataset/sources/__init__.py +45 -0
  28. agent_eval/dataset/sources/llm_generator.py +219 -0
  29. agent_eval/dataset/sources/manual.py +172 -0
  30. agent_eval/dataset/sources/regression.py +201 -0
  31. agent_eval/dataset/sources/trace_mining.py +277 -0
  32. agent_eval/dataset/storage.py +342 -0
  33. agent_eval/dataset/version.py +72 -0
  34. agent_eval/examples/__init__.py +0 -0
  35. agent_eval/examples/basic_usage.py +175 -0
  36. agent_eval/examples/mock_runner.py +195 -0
  37. agent_eval/graders/__init__.py +91 -0
  38. agent_eval/graders/artifact_check.py +114 -0
  39. agent_eval/graders/code_based.py +101 -0
  40. agent_eval/graders/human.py +77 -0
  41. agent_eval/graders/metric.py +142 -0
  42. agent_eval/graders/model_based.py +179 -0
  43. agent_eval/graders/state_check.py +106 -0
  44. agent_eval/graders/step_level.py +116 -0
  45. agent_eval/graders/tool_calls.py +102 -0
  46. agent_eval/graders/transcript.py +86 -0
  47. agent_eval/metrics/__init__.py +110 -0
  48. agent_eval/metrics/answer_relevancy.py +57 -0
  49. agent_eval/metrics/base.py +155 -0
  50. agent_eval/metrics/batch_evaluation.py +267 -0
  51. agent_eval/metrics/context_precision.py +62 -0
  52. agent_eval/metrics/context_recall.py +71 -0
  53. agent_eval/metrics/faithfulness.py +72 -0
  54. agent_eval/metrics/llm_judge.py +100 -0
  55. agent_eval/metrics/prompt_metric.py +150 -0
  56. agent_eval/metrics/pytest_plugin.py +308 -0
  57. agent_eval/metrics/report.py +149 -0
  58. agent_eval/metrics/synthetic_data.py +203 -0
  59. agent_eval/storage/__init__.py +17 -0
  60. agent_eval/storage/memory.py +95 -0
  61. agent_eval/storage/sqlite.py +240 -0
  62. agent_eval/trace/__init__.py +16 -0
  63. agent_eval/trace/phoenix.py +144 -0
@@ -0,0 +1,95 @@
1
+ """
2
+ In-memory storage implementation.
3
+
4
+ Useful for testing and short-lived runs.
5
+ Data is lost when the process exits.
6
+
7
+ Usage:
8
+ storage = MemoryStorage()
9
+ await storage.save_run(run_result)
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import copy
15
+ from typing import Any
16
+
17
+ from agent_eval.core.types import EvalSuite, RunResult
18
+ from agent_eval.dataset.storage import MemoryDatasetStorage
19
+
20
+
21
+ class MemoryStorage:
22
+ """内存存储实现 — 用于测试"""
23
+
24
+ def __init__(self):
25
+ self._runs: dict[str, RunResult] = {}
26
+ self._suites: dict[str, EvalSuite] = {}
27
+ self._human_score_requests: list[dict[str, Any]] = []
28
+ # 数据集存储组合暴露 (与 runs/suites 分表, 见 dataset/storage.py)
29
+ self.datasets = MemoryDatasetStorage()
30
+
31
+ # ── Run 操作 ──
32
+
33
+ async def save_run(self, run: RunResult) -> None:
34
+ """保存运行结果"""
35
+ self._runs[run.run_id] = copy.deepcopy(run)
36
+
37
+ async def get_run(self, run_id: str) -> RunResult | None:
38
+ """获取运行结果"""
39
+ run = self._runs.get(run_id)
40
+ return copy.deepcopy(run) if run else None
41
+
42
+ async def list_runs(
43
+ self, suite_name: str | None = None, limit: int = 50
44
+ ) -> list[RunResult]:
45
+ """列出运行历史"""
46
+ runs = list(self._runs.values())
47
+ if suite_name:
48
+ runs = [r for r in runs if r.suite_name == suite_name]
49
+ runs.sort(key=lambda r: r.started_at, reverse=True)
50
+ return runs[:limit]
51
+
52
+ async def delete_run(self, run_id: str) -> bool:
53
+ """删除运行结果"""
54
+ if run_id in self._runs:
55
+ del self._runs[run_id]
56
+ return True
57
+ return False
58
+
59
+ # ── Suite 操作 ──
60
+
61
+ async def save_suite(self, suite: EvalSuite) -> None:
62
+ """保存评测套件"""
63
+ self._suites[suite.name] = copy.deepcopy(suite)
64
+
65
+ async def get_suite(self, name: str) -> EvalSuite | None:
66
+ """获取评测套件"""
67
+ suite = self._suites.get(name)
68
+ return copy.deepcopy(suite) if suite else None
69
+
70
+ async def list_suites(self) -> list[EvalSuite]:
71
+ """列出所有评测套件"""
72
+ return list(self._suites.values())
73
+
74
+ async def delete_suite(self, name: str) -> bool:
75
+ """删除评测套件"""
76
+ if name in self._suites:
77
+ del self._suites[name]
78
+ return True
79
+ return False
80
+
81
+ # ── 人工评分请求 ──
82
+
83
+ async def save_human_score_request(self, request: dict[str, Any]) -> None:
84
+ """保存人工评分请求 (HumanGrader pending 语义)"""
85
+ self._human_score_requests.append(dict(request))
86
+
87
+ async def list_human_score_requests(
88
+ self, run_id: str | None = None
89
+ ) -> list[dict[str, Any]]:
90
+ """列出人工评分请求"""
91
+ if run_id is None:
92
+ return [dict(r) for r in self._human_score_requests]
93
+ return [
94
+ dict(r) for r in self._human_score_requests if r.get("run_id") == run_id
95
+ ]
@@ -0,0 +1,240 @@
1
+ """
2
+ SQLite storage implementation.
3
+
4
+ Default storage backend for single-machine use.
5
+ No external services required.
6
+
7
+ Usage:
8
+ storage = SqliteStorage(db_path="./aeval.db")
9
+ await storage.initialize()
10
+ await storage.save_run(run_result)
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import time
17
+ from typing import Any
18
+
19
+ import aiosqlite
20
+
21
+ from agent_eval.core.types import EvalSuite, RunResult
22
+ from agent_eval.dataset.storage import SqliteDatasetStorage
23
+
24
+
25
+ class SqliteStorage:
26
+ """
27
+ SQLite 存储实现。
28
+
29
+ 适合单机开发 / 轻量使用。
30
+ 无需额外服务, 开箱即用。
31
+ """
32
+
33
+ def __init__(self, db_path: str = "./aeval.db"):
34
+ """
35
+ Args:
36
+ db_path: SQLite 数据库文件路径
37
+ """
38
+ self.db_path = db_path
39
+ self._initialized = False
40
+ # 数据集存储组合暴露 (同一 db 文件, datasets/dataset_items 分表)
41
+ self.datasets = SqliteDatasetStorage(db_path)
42
+
43
+ async def initialize(self) -> None:
44
+ """初始化数据库表结构"""
45
+ async with aiosqlite.connect(self.db_path) as db:
46
+ await db.executescript("""
47
+ CREATE TABLE IF NOT EXISTS runs (
48
+ run_id TEXT PRIMARY KEY,
49
+ suite_name TEXT,
50
+ status TEXT,
51
+ started_at REAL,
52
+ completed_at REAL,
53
+ data TEXT
54
+ );
55
+
56
+ CREATE TABLE IF NOT EXISTS suites (
57
+ name TEXT PRIMARY KEY,
58
+ data TEXT,
59
+ created_at REAL
60
+ );
61
+
62
+ CREATE TABLE IF NOT EXISTS human_score_requests (
63
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
64
+ run_id TEXT,
65
+ task_id TEXT,
66
+ trial_index INTEGER,
67
+ grader_name TEXT,
68
+ data TEXT,
69
+ created_at REAL
70
+ );
71
+
72
+ CREATE INDEX IF NOT EXISTS idx_runs_suite
73
+ ON runs(suite_name);
74
+ CREATE INDEX IF NOT EXISTS idx_runs_status
75
+ ON runs(status);
76
+ CREATE INDEX IF NOT EXISTS idx_runs_started
77
+ ON runs(started_at);
78
+ CREATE INDEX IF NOT EXISTS idx_human_requests_run
79
+ ON human_score_requests(run_id);
80
+ """)
81
+ await self.datasets.initialize()
82
+ self._initialized = True
83
+
84
+ def _ensure_initialized(self) -> None:
85
+ if not self._initialized:
86
+ raise RuntimeError(
87
+ "SqliteStorage not initialized. Call await storage.initialize() first."
88
+ )
89
+
90
+ # ── Run 操作 ──
91
+
92
+ async def save_run(self, run: RunResult) -> None:
93
+ """保存运行结果"""
94
+ self._ensure_initialized()
95
+ async with aiosqlite.connect(self.db_path) as db:
96
+ await db.execute(
97
+ """INSERT OR REPLACE INTO runs
98
+ (run_id, suite_name, status, started_at, completed_at, data)
99
+ VALUES (?, ?, ?, ?, ?, ?)""",
100
+ (
101
+ run.run_id,
102
+ run.suite_name,
103
+ run.status,
104
+ run.started_at,
105
+ run.completed_at,
106
+ json.dumps(run.model_dump(), default=str),
107
+ ),
108
+ )
109
+ await db.commit()
110
+
111
+ async def get_run(self, run_id: str) -> RunResult | None:
112
+ """获取运行结果"""
113
+ self._ensure_initialized()
114
+ async with aiosqlite.connect(self.db_path) as db:
115
+ db.row_factory = aiosqlite.Row
116
+ cursor = await db.execute(
117
+ "SELECT data FROM runs WHERE run_id = ?", (run_id,)
118
+ )
119
+ row = await cursor.fetchone()
120
+ if row:
121
+ data = json.loads(row["data"])
122
+ return RunResult(**data)
123
+ return None
124
+
125
+ async def list_runs(
126
+ self, suite_name: str | None = None, limit: int = 50
127
+ ) -> list[RunResult]:
128
+ """列出运行历史"""
129
+ self._ensure_initialized()
130
+ async with aiosqlite.connect(self.db_path) as db:
131
+ db.row_factory = aiosqlite.Row
132
+ if suite_name:
133
+ cursor = await db.execute(
134
+ "SELECT data FROM runs WHERE suite_name = ? ORDER BY started_at DESC LIMIT ?",
135
+ (suite_name, limit),
136
+ )
137
+ else:
138
+ cursor = await db.execute(
139
+ "SELECT data FROM runs ORDER BY started_at DESC LIMIT ?",
140
+ (limit,),
141
+ )
142
+ rows = await cursor.fetchall()
143
+ return [RunResult(**json.loads(r["data"])) for r in rows]
144
+
145
+ async def delete_run(self, run_id: str) -> bool:
146
+ """删除运行结果"""
147
+ self._ensure_initialized()
148
+ async with aiosqlite.connect(self.db_path) as db:
149
+ cursor = await db.execute(
150
+ "DELETE FROM runs WHERE run_id = ?", (run_id,)
151
+ )
152
+ await db.commit()
153
+ return cursor.rowcount > 0
154
+
155
+ # ── Suite 操作 ──
156
+
157
+ async def save_suite(self, suite: EvalSuite) -> None:
158
+ """保存评测套件"""
159
+ self._ensure_initialized()
160
+ async with aiosqlite.connect(self.db_path) as db:
161
+ await db.execute(
162
+ """INSERT OR REPLACE INTO suites (name, data, created_at)
163
+ VALUES (?, ?, ?)""",
164
+ (suite.name, json.dumps(suite.model_dump(), default=str), time.time()),
165
+ )
166
+ await db.commit()
167
+
168
+ async def get_suite(self, name: str) -> EvalSuite | None:
169
+ """获取评测套件"""
170
+ self._ensure_initialized()
171
+ async with aiosqlite.connect(self.db_path) as db:
172
+ db.row_factory = aiosqlite.Row
173
+ cursor = await db.execute(
174
+ "SELECT data FROM suites WHERE name = ?", (name,)
175
+ )
176
+ row = await cursor.fetchone()
177
+ if row:
178
+ data = json.loads(row["data"])
179
+ return EvalSuite(**data)
180
+ return None
181
+
182
+ async def list_suites(self) -> list[EvalSuite]:
183
+ """列出所有评测套件"""
184
+ self._ensure_initialized()
185
+ async with aiosqlite.connect(self.db_path) as db:
186
+ db.row_factory = aiosqlite.Row
187
+ cursor = await db.execute("SELECT data FROM suites ORDER BY created_at DESC")
188
+ rows = await cursor.fetchall()
189
+ return [EvalSuite(**json.loads(r["data"])) for r in rows]
190
+
191
+ async def delete_suite(self, name: str) -> bool:
192
+ """删除评测套件"""
193
+ self._ensure_initialized()
194
+ async with aiosqlite.connect(self.db_path) as db:
195
+ cursor = await db.execute(
196
+ "DELETE FROM suites WHERE name = ?", (name,)
197
+ )
198
+ await db.commit()
199
+ return cursor.rowcount > 0
200
+
201
+ # ── 人工评分请求 ──
202
+
203
+ async def save_human_score_request(self, request: dict[str, Any]) -> None:
204
+ """保存人工评分请求 (HumanGrader pending 语义)"""
205
+ self._ensure_initialized()
206
+ async with aiosqlite.connect(self.db_path) as db:
207
+ await db.execute(
208
+ """INSERT INTO human_score_requests
209
+ (run_id, task_id, trial_index, grader_name, data, created_at)
210
+ VALUES (?, ?, ?, ?, ?, ?)""",
211
+ (
212
+ request.get("run_id", ""),
213
+ request.get("task_id", ""),
214
+ request.get("trial_index", 0),
215
+ request.get("grader_name", "human"),
216
+ json.dumps(request, default=str),
217
+ time.time(),
218
+ ),
219
+ )
220
+ await db.commit()
221
+
222
+ async def list_human_score_requests(
223
+ self, run_id: str | None = None
224
+ ) -> list[dict[str, Any]]:
225
+ """列出人工评分请求"""
226
+ self._ensure_initialized()
227
+ async with aiosqlite.connect(self.db_path) as db:
228
+ db.row_factory = aiosqlite.Row
229
+ if run_id is not None:
230
+ cursor = await db.execute(
231
+ "SELECT data FROM human_score_requests WHERE run_id = ? "
232
+ "ORDER BY id ASC",
233
+ (run_id,),
234
+ )
235
+ else:
236
+ cursor = await db.execute(
237
+ "SELECT data FROM human_score_requests ORDER BY id ASC"
238
+ )
239
+ rows = await cursor.fetchall()
240
+ return [json.loads(r["data"]) for r in rows]
@@ -0,0 +1,16 @@
1
+ """
2
+ Trace provider implementations for the Aeval evaluation framework.
3
+
4
+ Provides a Phoenix-based trace provider out of the box.
5
+ Other backends (Jaeger, Tempo, etc.) can be implemented by users.
6
+
7
+ Usage:
8
+ from agent_eval.trace import PhoenixProvider
9
+
10
+ provider = PhoenixProvider(endpoint="http://localhost:6006")
11
+ spans = await provider.get_spans("trace_abc123")
12
+ """
13
+
14
+ from agent_eval.trace.phoenix import PhoenixProvider
15
+
16
+ __all__ = ["PhoenixProvider"]
@@ -0,0 +1,144 @@
1
+ """
2
+ Phoenix trace provider implementation.
3
+
4
+ Fetches trace spans from an Arize Phoenix instance.
5
+ Requires: pip install arize-phoenix
6
+
7
+ Usage:
8
+ provider = PhoenixProvider(endpoint="http://localhost:6006")
9
+ spans = await provider.get_spans("trace_abc123")
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from typing import Any
15
+
16
+
17
+ class PhoenixProvider:
18
+ """
19
+ Phoenix TraceProvider 实现。
20
+
21
+ 通过 Phoenix Python SDK 获取 trace span 数据。
22
+ """
23
+
24
+ def __init__(
25
+ self,
26
+ endpoint: str = "http://localhost:6006",
27
+ project: str = "default",
28
+ ):
29
+ """
30
+ Args:
31
+ endpoint: Phoenix UI endpoint (e.g., http://localhost:6006)
32
+ project: Phoenix project name
33
+ """
34
+ self.endpoint = endpoint
35
+ self.project = project
36
+ self._client = None
37
+
38
+ def _get_client(self):
39
+ """Lazy-init Phoenix client"""
40
+ if self._client is None:
41
+ try:
42
+ # phoenix >= 5 移除了顶层 px.Client, 统一走 phoenix.client;
43
+ # base_url 在旧版叫 endpoint, 新 API 只认 base_url。
44
+ from phoenix.client import Client as PhoenixClient
45
+
46
+ self._client = PhoenixClient(base_url=self.endpoint)
47
+ except ImportError as e:
48
+ raise RuntimeError(
49
+ "phoenix package not installed. "
50
+ "Install with: pip install arize-phoenix"
51
+ ) from e
52
+ return self._client
53
+
54
+ async def get_spans(self, trace_id: str) -> list[dict[str, Any]]:
55
+ """
56
+ 获取一个 trace 的所有 span。
57
+
58
+ Args:
59
+ trace_id: OTel trace ID
60
+
61
+ Returns:
62
+ 标准化的 span 列表
63
+ """
64
+ import asyncio
65
+
66
+ # Phoenix SDK 是同步的, 在线程中运行
67
+ return await asyncio.to_thread(self._get_spans_sync, trace_id)
68
+
69
+ def _get_spans_sync(self, trace_id: str) -> list[dict[str, Any]]:
70
+ """同步获取 spans"""
71
+ client = self._get_client()
72
+
73
+ try:
74
+ df = client.spans.get_spans_dataframe(project_name=self.project)
75
+ except Exception:
76
+ return []
77
+
78
+ if df is None or df.empty:
79
+ return []
80
+
81
+ # 过滤指定 trace
82
+ trace_col = "context.trace_id"
83
+ if trace_col not in df.columns:
84
+ return []
85
+
86
+ trace_spans = df[df[trace_col] == trace_id]
87
+ if trace_spans.empty:
88
+ return []
89
+
90
+ # 转换为标准格式
91
+ return self._normalize_spans(trace_spans.to_dict("records"))
92
+
93
+ async def get_trace_ids(
94
+ self,
95
+ filters: dict[str, Any] | None = None,
96
+ limit: int = 100,
97
+ ) -> list[str]:
98
+ """
99
+ 查询 trace ID 列表。
100
+
101
+ Args:
102
+ filters: 过滤条件 (暂未实现, 保留接口)
103
+ limit: 返回数量限制
104
+
105
+ Returns:
106
+ trace ID 列表
107
+ """
108
+ import asyncio
109
+
110
+ return await asyncio.to_thread(self._get_trace_ids_sync, limit)
111
+
112
+ def _get_trace_ids_sync(self, limit: int) -> list[str]:
113
+ """同步获取 trace IDs"""
114
+ client = self._get_client()
115
+
116
+ try:
117
+ df = client.spans.get_spans_dataframe(project_name=self.project)
118
+ except Exception:
119
+ return []
120
+
121
+ if df is None or df.empty:
122
+ return []
123
+
124
+ trace_col = "context.trace_id"
125
+ if trace_col not in df.columns:
126
+ return []
127
+
128
+ trace_ids = df[trace_col].unique().tolist()
129
+ return trace_ids[:limit]
130
+
131
+ def _normalize_spans(self, raw_spans: list[dict[str, Any]]) -> list[dict[str, Any]]:
132
+ """将 Phoenix DataFrame 格式归一化为标准格式"""
133
+ normalized = []
134
+ for span in raw_spans:
135
+ normalized.append({
136
+ "name": span.get("name", ""),
137
+ "attributes": span.get("attributes", {}),
138
+ "start_time": str(span.get("start_time", "")),
139
+ "end_time": str(span.get("end_time", "")),
140
+ "status": span.get("status", {}),
141
+ "trace_id": span.get("context.trace_id", ""),
142
+ "span_id": span.get("context.span_id", ""),
143
+ })
144
+ return normalized