aeval-framework 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- aeval_framework-0.1.0.dist-info/METADATA +42 -0
- aeval_framework-0.1.0.dist-info/RECORD +63 -0
- aeval_framework-0.1.0.dist-info/WHEEL +4 -0
- aeval_framework-0.1.0.dist-info/entry_points.txt +2 -0
- agent_eval/__init__.py +14 -0
- agent_eval/api/__init__.py +14 -0
- agent_eval/api/app.py +82 -0
- agent_eval/api/events.py +96 -0
- agent_eval/api/routes/__init__.py +0 -0
- agent_eval/api/routes/datasets.py +441 -0
- agent_eval/api/routes/graders.py +19 -0
- agent_eval/api/routes/metrics.py +49 -0
- agent_eval/api/routes/runs.py +573 -0
- agent_eval/api/routes/suites.py +84 -0
- agent_eval/api/routes/tasks.py +114 -0
- agent_eval/api/standalone.py +105 -0
- agent_eval/cli.py +455 -0
- agent_eval/core/__init__.py +48 -0
- agent_eval/core/contract.py +296 -0
- agent_eval/core/metrics.py +184 -0
- agent_eval/core/runner.py +868 -0
- agent_eval/core/suite.py +60 -0
- agent_eval/core/types.py +227 -0
- agent_eval/dataset/__init__.py +31 -0
- agent_eval/dataset/models.py +199 -0
- agent_eval/dataset/quality.py +194 -0
- agent_eval/dataset/sources/__init__.py +45 -0
- agent_eval/dataset/sources/llm_generator.py +219 -0
- agent_eval/dataset/sources/manual.py +172 -0
- agent_eval/dataset/sources/regression.py +201 -0
- agent_eval/dataset/sources/trace_mining.py +277 -0
- agent_eval/dataset/storage.py +342 -0
- agent_eval/dataset/version.py +72 -0
- agent_eval/examples/__init__.py +0 -0
- agent_eval/examples/basic_usage.py +175 -0
- agent_eval/examples/mock_runner.py +195 -0
- agent_eval/graders/__init__.py +91 -0
- agent_eval/graders/artifact_check.py +114 -0
- agent_eval/graders/code_based.py +101 -0
- agent_eval/graders/human.py +77 -0
- agent_eval/graders/metric.py +142 -0
- agent_eval/graders/model_based.py +179 -0
- agent_eval/graders/state_check.py +106 -0
- agent_eval/graders/step_level.py +116 -0
- agent_eval/graders/tool_calls.py +102 -0
- agent_eval/graders/transcript.py +86 -0
- agent_eval/metrics/__init__.py +110 -0
- agent_eval/metrics/answer_relevancy.py +57 -0
- agent_eval/metrics/base.py +155 -0
- agent_eval/metrics/batch_evaluation.py +267 -0
- agent_eval/metrics/context_precision.py +62 -0
- agent_eval/metrics/context_recall.py +71 -0
- agent_eval/metrics/faithfulness.py +72 -0
- agent_eval/metrics/llm_judge.py +100 -0
- agent_eval/metrics/prompt_metric.py +150 -0
- agent_eval/metrics/pytest_plugin.py +308 -0
- agent_eval/metrics/report.py +149 -0
- agent_eval/metrics/synthetic_data.py +203 -0
- agent_eval/storage/__init__.py +17 -0
- agent_eval/storage/memory.py +95 -0
- agent_eval/storage/sqlite.py +240 -0
- agent_eval/trace/__init__.py +16 -0
- agent_eval/trace/phoenix.py +144 -0
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""Regression sample extraction — close the eval→failure→extract→re-eval loop.
|
|
2
|
+
|
|
3
|
+
Extracts failed trials from a Run into regression items (D5):
|
|
4
|
+
- one item per failing TASK (first failed trial provides trace_id reference)
|
|
5
|
+
- prompt from the trial transcript's first message, falling back to the
|
|
6
|
+
suite task's prompt; trials without any prompt source are skipped
|
|
7
|
+
- graders/env copied from the suite task so extracted items are re-runnable
|
|
8
|
+
- merging into a dataset dedupes on normalized prompt so regression samples
|
|
9
|
+
don't balloon with repeated runs
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import re
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
from agent_eval.core.types import EvalSuite, RunResult, TrialResult
|
|
19
|
+
from agent_eval.dataset.models import EvalDataset, EvalDatasetItem, SourceType, now_ms
|
|
20
|
+
|
|
21
|
+
# 回归样本数上限默认值 (D5)
|
|
22
|
+
DEFAULT_MAX_ITEMS = 50
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def normalize_prompt(prompt: str) -> str:
|
|
26
|
+
"""prompt 归一化 (去首尾空白 + 折叠连续空白), 用于去重比较"""
|
|
27
|
+
return re.sub(r"\s+", " ", prompt or "").strip()
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class RegressionReport:
|
|
32
|
+
"""提取/合入报告"""
|
|
33
|
+
|
|
34
|
+
run_id: str = ""
|
|
35
|
+
failed_trials: int = 0
|
|
36
|
+
extracted: int = 0
|
|
37
|
+
skipped: list[dict[str, Any]] = field(default_factory=list)
|
|
38
|
+
merged: int = 0
|
|
39
|
+
merged_skipped: list[dict[str, Any]] = field(default_factory=list)
|
|
40
|
+
|
|
41
|
+
def to_dict(self) -> dict[str, Any]:
|
|
42
|
+
return {
|
|
43
|
+
"run_id": self.run_id,
|
|
44
|
+
"failed_trials": self.failed_trials,
|
|
45
|
+
"extracted": self.extracted,
|
|
46
|
+
"skipped": self.skipped,
|
|
47
|
+
"merged": self.merged,
|
|
48
|
+
"merged_skipped": self.merged_skipped,
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class RegressionExtractor:
|
|
53
|
+
"""从 Run 失败 trial 中提取回归样本"""
|
|
54
|
+
|
|
55
|
+
def __init__(self, max_items: int = DEFAULT_MAX_ITEMS):
|
|
56
|
+
self.max_items = max_items
|
|
57
|
+
|
|
58
|
+
def extract_from_run(
|
|
59
|
+
self,
|
|
60
|
+
run: RunResult,
|
|
61
|
+
suite: EvalSuite | None = None,
|
|
62
|
+
max_items: int | None = None,
|
|
63
|
+
) -> tuple[list[EvalDatasetItem], RegressionReport]:
|
|
64
|
+
"""
|
|
65
|
+
从一次 Run 的失败 trial 中提取回归条目。
|
|
66
|
+
|
|
67
|
+
同一 task 的多个失败 trial 只提取一条 (prompt 相同,
|
|
68
|
+
取首个失败 trial 的 trace_id 为引用)。
|
|
69
|
+
|
|
70
|
+
Args:
|
|
71
|
+
run: 评测运行结果
|
|
72
|
+
suite: 可选的源 Suite — 提供任务 prompt 兜底与 graders/env 复用
|
|
73
|
+
max_items: 本次提取上限 (默认取实例配置)
|
|
74
|
+
|
|
75
|
+
Returns:
|
|
76
|
+
(items, report)
|
|
77
|
+
"""
|
|
78
|
+
cap = max_items if max_items is not None else self.max_items
|
|
79
|
+
suite_tasks = {t.id: t for t in suite.tasks} if suite else {}
|
|
80
|
+
|
|
81
|
+
report = RegressionReport(run_id=run.run_id)
|
|
82
|
+
items: list[EvalDatasetItem] = []
|
|
83
|
+
|
|
84
|
+
for task_id, trials in run.trials.items():
|
|
85
|
+
failed = [t for t in trials if not t.success]
|
|
86
|
+
report.failed_trials += len(failed)
|
|
87
|
+
if not failed:
|
|
88
|
+
continue
|
|
89
|
+
|
|
90
|
+
# 同一 task 的多个失败 trial 只提取一条 (prompt 相同, 取首个失败 trial)
|
|
91
|
+
first_failure: TrialResult = failed[0]
|
|
92
|
+
prompt = self._prompt_of(first_failure, suite_tasks.get(task_id))
|
|
93
|
+
if not prompt:
|
|
94
|
+
report.skipped.append({
|
|
95
|
+
"task_id": task_id,
|
|
96
|
+
"trial_index": first_failure.trial_index,
|
|
97
|
+
"reason": "no prompt available (empty transcript, no suite task)",
|
|
98
|
+
})
|
|
99
|
+
continue
|
|
100
|
+
|
|
101
|
+
if len(items) >= cap:
|
|
102
|
+
report.skipped.append({
|
|
103
|
+
"task_id": task_id,
|
|
104
|
+
"trial_index": first_failure.trial_index,
|
|
105
|
+
"reason": f"max_items={cap} reached",
|
|
106
|
+
})
|
|
107
|
+
continue
|
|
108
|
+
|
|
109
|
+
task = suite_tasks.get(task_id)
|
|
110
|
+
error_note = (
|
|
111
|
+
f" (error: {first_failure.error[:120]})"
|
|
112
|
+
if first_failure.error
|
|
113
|
+
else ""
|
|
114
|
+
)
|
|
115
|
+
items.append(
|
|
116
|
+
EvalDatasetItem(
|
|
117
|
+
id=f"regression_{task_id}_{first_failure.trial_index}",
|
|
118
|
+
prompt=prompt,
|
|
119
|
+
description=f"Regression: {task_id}{error_note}",
|
|
120
|
+
graders=list(task.graders) if task else [],
|
|
121
|
+
env=dict(task.env) if task else {},
|
|
122
|
+
metadata={
|
|
123
|
+
"capabilities": [],
|
|
124
|
+
"regression": {
|
|
125
|
+
"run_id": run.run_id,
|
|
126
|
+
"task_id": task_id,
|
|
127
|
+
"trial_index": first_failure.trial_index,
|
|
128
|
+
"error": first_failure.error,
|
|
129
|
+
},
|
|
130
|
+
},
|
|
131
|
+
source_type=SourceType.REGRESSION,
|
|
132
|
+
source_ref=first_failure.trace_id or run.run_id,
|
|
133
|
+
created_at=now_ms(),
|
|
134
|
+
)
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
report.extracted = len(items)
|
|
138
|
+
return items, report
|
|
139
|
+
|
|
140
|
+
@staticmethod
|
|
141
|
+
def _prompt_of(trial: TrialResult, suite_task: Any = None) -> str:
|
|
142
|
+
"""trial prompt: transcript 首条消息优先, suite 任务 prompt 兜底"""
|
|
143
|
+
if trial.transcript:
|
|
144
|
+
first = trial.transcript[0]
|
|
145
|
+
if isinstance(first, dict):
|
|
146
|
+
content = first.get("content", "")
|
|
147
|
+
if isinstance(content, str) and content.strip():
|
|
148
|
+
return content
|
|
149
|
+
if suite_task is not None:
|
|
150
|
+
return suite_task.prompt or ""
|
|
151
|
+
return ""
|
|
152
|
+
|
|
153
|
+
def merge_into_dataset(
|
|
154
|
+
self,
|
|
155
|
+
dataset: EvalDataset,
|
|
156
|
+
items: list[EvalDatasetItem],
|
|
157
|
+
) -> tuple[EvalDataset, RegressionReport]:
|
|
158
|
+
"""
|
|
159
|
+
将提取的条目合入数据集 — 按 prompt 归一化去重。
|
|
160
|
+
|
|
161
|
+
数据集或新增条目中 prompt 归一化后重复的条目跳过 (记录于报告),
|
|
162
|
+
避免回归样本随 run 次数膨胀。
|
|
163
|
+
|
|
164
|
+
Returns:
|
|
165
|
+
(更新后的 dataset 副本, 报告)
|
|
166
|
+
"""
|
|
167
|
+
report = RegressionReport(run_id="")
|
|
168
|
+
existing = {normalize_prompt(i.prompt) for i in dataset.items if i.prompt}
|
|
169
|
+
|
|
170
|
+
merged_items = list(dataset.items)
|
|
171
|
+
for raw in items:
|
|
172
|
+
# 容忍 dict 形态 (API/JSON 透传场景)
|
|
173
|
+
item = raw if isinstance(raw, EvalDatasetItem) else EvalDatasetItem(**raw)
|
|
174
|
+
key = normalize_prompt(item.prompt)
|
|
175
|
+
if not key or key in existing:
|
|
176
|
+
report.merged_skipped.append({
|
|
177
|
+
"item_id": item.id,
|
|
178
|
+
"reason": "duplicate prompt (normalized) already in dataset",
|
|
179
|
+
})
|
|
180
|
+
continue
|
|
181
|
+
existing.add(key)
|
|
182
|
+
merged_items.append(item)
|
|
183
|
+
report.merged += 1
|
|
184
|
+
|
|
185
|
+
# 数据集内 ID 冲突时重命名 (不同 run 的同名回归条目)
|
|
186
|
+
taken_ids = set()
|
|
187
|
+
updated = []
|
|
188
|
+
for item in merged_items:
|
|
189
|
+
if item.id in taken_ids:
|
|
190
|
+
suffix = 1
|
|
191
|
+
while f"{item.id}_v{suffix}" in taken_ids:
|
|
192
|
+
suffix += 1
|
|
193
|
+
item = item.model_copy(update={"id": f"{item.id}_v{suffix}"})
|
|
194
|
+
taken_ids.add(item.id)
|
|
195
|
+
updated.append(item)
|
|
196
|
+
|
|
197
|
+
new_dataset = dataset.model_copy(update={
|
|
198
|
+
"items": updated,
|
|
199
|
+
"updated_at": now_ms(),
|
|
200
|
+
})
|
|
201
|
+
return new_dataset, report
|
|
@@ -0,0 +1,277 @@
|
|
|
1
|
+
"""Trace Mining source — mine eval items from real Agent traces.
|
|
2
|
+
|
|
3
|
+
Three first-version strategies (D4):
|
|
4
|
+
- failed_tasks: traces containing error spans
|
|
5
|
+
- long_running: traces whose duration exceeds P90 × multiplier
|
|
6
|
+
- diverse_sampling: deterministic hash-based uniform sampling
|
|
7
|
+
|
|
8
|
+
The prompt comes from the ROOT span's input attribute; traces without one are
|
|
9
|
+
counted as `skipped` in the mining report (never guessed). Every mined item
|
|
10
|
+
keeps trace provenance (source_type=trace_mining, source_ref=trace_id).
|
|
11
|
+
`user_dissatisfied` stays as an enum placeholder until a user-feedback
|
|
12
|
+
data channel exists (design §18, first-version scope note).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import hashlib
|
|
18
|
+
from dataclasses import dataclass, field
|
|
19
|
+
from enum import Enum
|
|
20
|
+
from typing import Any
|
|
21
|
+
|
|
22
|
+
from agent_eval.core.contract import TraceProvider
|
|
23
|
+
from agent_eval.dataset.models import EvalDatasetItem, SourceType, now_ms
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class MiningStrategy(str, Enum):
|
|
27
|
+
"""挖掘策略 (首版 3 个; user_dissatisfied 依赖用户反馈通道, 留枚举位)"""
|
|
28
|
+
|
|
29
|
+
FAILED_TASKS = "failed_tasks"
|
|
30
|
+
LONG_RUNNING = "long_running"
|
|
31
|
+
DIVERSE_SAMPLING = "diverse_sampling"
|
|
32
|
+
USER_DISSATISFIED = "user_dissatisfied" # 首版未实现
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
# 根 span input 属性的候选键 (Phoenix/OpenInference 惯例优先)
|
|
36
|
+
_INPUT_ATTR_KEYS = ("input.value", "input", "agent.input", "llm.input_messages")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class TraceMiner:
|
|
40
|
+
"""从 TraceProvider 提供的 trace 中按策略挖掘评测条目"""
|
|
41
|
+
|
|
42
|
+
def __init__(
|
|
43
|
+
self,
|
|
44
|
+
trace_provider: TraceProvider,
|
|
45
|
+
long_running_multiplier: float = 2.0,
|
|
46
|
+
):
|
|
47
|
+
"""
|
|
48
|
+
Args:
|
|
49
|
+
trace_provider: trace 数据源 (get_trace_ids + get_spans)
|
|
50
|
+
long_running_multiplier: long_running 策略的 P90 倍数阈值
|
|
51
|
+
"""
|
|
52
|
+
self.trace_provider = trace_provider
|
|
53
|
+
self.long_running_multiplier = max(1.0, long_running_multiplier)
|
|
54
|
+
|
|
55
|
+
async def mine(
|
|
56
|
+
self,
|
|
57
|
+
strategy: str | MiningStrategy,
|
|
58
|
+
filters: dict[str, Any] | None = None,
|
|
59
|
+
limit: int = 20,
|
|
60
|
+
candidate_limit: int = 100,
|
|
61
|
+
) -> MiningReport:
|
|
62
|
+
"""
|
|
63
|
+
执行挖掘。
|
|
64
|
+
|
|
65
|
+
Args:
|
|
66
|
+
strategy: 挖掘策略名
|
|
67
|
+
filters: 传给 get_trace_ids 的过滤条件
|
|
68
|
+
limit: 最多产出的条目数
|
|
69
|
+
candidate_limit: 最多检查的候选 trace 数 (先过滤再拉 spans, 控制成本)
|
|
70
|
+
|
|
71
|
+
Returns:
|
|
72
|
+
MiningReport: items + skipped + 统计
|
|
73
|
+
"""
|
|
74
|
+
strategy_enum = MiningStrategy(strategy)
|
|
75
|
+
if strategy_enum == MiningStrategy.USER_DISSATISFIED:
|
|
76
|
+
raise NotImplementedError(
|
|
77
|
+
"user_dissatisfied mining requires a user-feedback data channel "
|
|
78
|
+
"(not available in the first version — see design §18 scope)"
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
trace_ids = await self.trace_provider.get_trace_ids(
|
|
82
|
+
filters=filters, limit=candidate_limit
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
# 先拉 spans, 再按策略筛选 (策略内限制候选数, 避免全量展开)
|
|
86
|
+
candidates: list[tuple[str, list[dict[str, Any]]]] = []
|
|
87
|
+
for trace_id in trace_ids:
|
|
88
|
+
spans = await self.trace_provider.get_spans(trace_id)
|
|
89
|
+
if spans:
|
|
90
|
+
candidates.append((trace_id, spans))
|
|
91
|
+
|
|
92
|
+
selected = self._select_by_strategy(strategy_enum, candidates, limit)
|
|
93
|
+
|
|
94
|
+
items: list[EvalDatasetItem] = []
|
|
95
|
+
skipped: list[dict[str, Any]] = []
|
|
96
|
+
for trace_id, spans in selected:
|
|
97
|
+
root = self._root_span(spans)
|
|
98
|
+
prompt = self._extract_prompt(root) if root is not None else None
|
|
99
|
+
if not prompt:
|
|
100
|
+
skipped.append({
|
|
101
|
+
"trace_id": trace_id,
|
|
102
|
+
"reason": "no user input found in root span",
|
|
103
|
+
})
|
|
104
|
+
continue
|
|
105
|
+
items.append(self._trace_to_item(trace_id, spans, strategy_enum, prompt))
|
|
106
|
+
|
|
107
|
+
return MiningReport(
|
|
108
|
+
strategy=strategy_enum.value,
|
|
109
|
+
candidates=len(trace_ids),
|
|
110
|
+
inspected=len(candidates),
|
|
111
|
+
items=items,
|
|
112
|
+
skipped=skipped,
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
# ── 策略筛选 ──
|
|
116
|
+
|
|
117
|
+
def _select_by_strategy(
|
|
118
|
+
self,
|
|
119
|
+
strategy: MiningStrategy,
|
|
120
|
+
candidates: list[tuple[str, list[dict[str, Any]]]],
|
|
121
|
+
limit: int,
|
|
122
|
+
) -> list[tuple[str, list[dict[str, Any]]]]:
|
|
123
|
+
if strategy == MiningStrategy.FAILED_TASKS:
|
|
124
|
+
selected = [(tid, spans) for tid, spans in candidates if self._is_failed(spans)]
|
|
125
|
+
elif strategy == MiningStrategy.LONG_RUNNING:
|
|
126
|
+
selected = self._select_long_running(candidates)
|
|
127
|
+
else: # DIVERSE_SAMPLING
|
|
128
|
+
selected = self._select_diverse([tid for tid, _ in candidates], limit)
|
|
129
|
+
selected = [(tid, spans) for tid, spans in candidates if tid in selected]
|
|
130
|
+
|
|
131
|
+
return selected[:limit]
|
|
132
|
+
|
|
133
|
+
def _select_long_running(
|
|
134
|
+
self,
|
|
135
|
+
candidates: list[tuple[str, list[dict[str, Any]]]],
|
|
136
|
+
) -> list[tuple[str, list[dict[str, Any]]]]:
|
|
137
|
+
timed = [
|
|
138
|
+
(tid, spans, self._trace_duration(spans))
|
|
139
|
+
for tid, spans in candidates
|
|
140
|
+
]
|
|
141
|
+
timed = [(tid, spans, d) for tid, spans, d in timed if d is not None]
|
|
142
|
+
if not timed:
|
|
143
|
+
return []
|
|
144
|
+
ordered = sorted(d for _, _, d in timed)
|
|
145
|
+
p90 = ordered[max(0, int(len(ordered) * 0.9) - 1)] if len(ordered) > 1 else ordered[0]
|
|
146
|
+
threshold = p90 * self.long_running_multiplier
|
|
147
|
+
return [(tid, spans) for tid, spans, d in timed if d > threshold]
|
|
148
|
+
|
|
149
|
+
def _select_diverse(self, trace_ids: list[str], limit: int) -> list[str]:
|
|
150
|
+
"""按 trace_id 哈希排序均匀采样 (确定性, 与到达顺序无关)"""
|
|
151
|
+
ranked = sorted(trace_ids, key=lambda tid: hashlib.md5(tid.encode()).hexdigest())
|
|
152
|
+
return ranked[:limit]
|
|
153
|
+
|
|
154
|
+
# ── Trace 分析 ──
|
|
155
|
+
|
|
156
|
+
@staticmethod
|
|
157
|
+
def _is_failed(spans: list[dict[str, Any]]) -> bool:
|
|
158
|
+
for span in spans:
|
|
159
|
+
status = span.get("status") or {}
|
|
160
|
+
status_code = (
|
|
161
|
+
status.get("status_code")
|
|
162
|
+
if isinstance(status, dict)
|
|
163
|
+
else str(status)
|
|
164
|
+
)
|
|
165
|
+
if str(status_code).upper() in ("ERROR", "STATUS_CODE_ERROR", "2"):
|
|
166
|
+
return True
|
|
167
|
+
attrs = span.get("attributes") or {}
|
|
168
|
+
if attrs.get("error") is True or attrs.get("error.type"):
|
|
169
|
+
return True
|
|
170
|
+
return False
|
|
171
|
+
|
|
172
|
+
@staticmethod
|
|
173
|
+
def _trace_duration(spans: list[dict[str, Any]]) -> float | None:
|
|
174
|
+
"""trace 时长 (span start/end 解析失败返回 None)"""
|
|
175
|
+
starts: list[float] = []
|
|
176
|
+
ends: list[float] = []
|
|
177
|
+
for span in spans:
|
|
178
|
+
s = TraceMiner._parse_time(span.get("start_time"))
|
|
179
|
+
e = TraceMiner._parse_time(span.get("end_time"))
|
|
180
|
+
if s is not None:
|
|
181
|
+
starts.append(s)
|
|
182
|
+
if e is not None:
|
|
183
|
+
ends.append(e)
|
|
184
|
+
if not starts or not ends:
|
|
185
|
+
return None
|
|
186
|
+
return max(0.0, max(ends) - min(starts))
|
|
187
|
+
|
|
188
|
+
@staticmethod
|
|
189
|
+
def _parse_time(value: Any) -> float | None:
|
|
190
|
+
if value is None or value == "":
|
|
191
|
+
return None
|
|
192
|
+
try:
|
|
193
|
+
return float(value)
|
|
194
|
+
except (TypeError, ValueError):
|
|
195
|
+
return None
|
|
196
|
+
|
|
197
|
+
@staticmethod
|
|
198
|
+
def _root_span(spans: list[dict[str, Any]]) -> dict[str, Any] | None:
|
|
199
|
+
"""根 span = 最早开始的 span (归一化 span 不含 parent_id)"""
|
|
200
|
+
if not spans:
|
|
201
|
+
return None
|
|
202
|
+
return min(
|
|
203
|
+
spans,
|
|
204
|
+
key=lambda s: TraceMiner._parse_time(s.get("start_time")) or 0.0,
|
|
205
|
+
)
|
|
206
|
+
|
|
207
|
+
@staticmethod
|
|
208
|
+
def _extract_prompt(root: dict[str, Any] | None) -> str | None:
|
|
209
|
+
"""从根 span input 属性提取用户输入; 缺失返回 None (不猜)"""
|
|
210
|
+
if root is None:
|
|
211
|
+
return None
|
|
212
|
+
attrs = root.get("attributes") or {}
|
|
213
|
+
for key in _INPUT_ATTR_KEYS:
|
|
214
|
+
value = attrs.get(key)
|
|
215
|
+
if isinstance(value, str) and value.strip():
|
|
216
|
+
return value.strip()
|
|
217
|
+
if isinstance(value, list) and value:
|
|
218
|
+
# llm.input_messages: [{"content": ...}] 形态兜底
|
|
219
|
+
first = value[0]
|
|
220
|
+
if isinstance(first, dict):
|
|
221
|
+
content = first.get("content")
|
|
222
|
+
if isinstance(content, str) and content.strip():
|
|
223
|
+
return content.strip()
|
|
224
|
+
return None
|
|
225
|
+
|
|
226
|
+
def _trace_to_item(
|
|
227
|
+
self,
|
|
228
|
+
trace_id: str,
|
|
229
|
+
spans: list[dict[str, Any]],
|
|
230
|
+
strategy: MiningStrategy,
|
|
231
|
+
prompt: str,
|
|
232
|
+
) -> EvalDatasetItem:
|
|
233
|
+
root = self._root_span(spans) or {}
|
|
234
|
+
failed = self._is_failed(spans)
|
|
235
|
+
description = (
|
|
236
|
+
f"Mined [{strategy.value}]: {root.get('name', 'unknown span')}"
|
|
237
|
+
+ (" (failed trace)" if failed else "")
|
|
238
|
+
)
|
|
239
|
+
return EvalDatasetItem(
|
|
240
|
+
id=f"mined_{strategy.value}_{trace_id[:12]}",
|
|
241
|
+
prompt=prompt,
|
|
242
|
+
description=description,
|
|
243
|
+
graders=[],
|
|
244
|
+
metadata={
|
|
245
|
+
"capabilities": [],
|
|
246
|
+
"mining": {
|
|
247
|
+
"strategy": strategy.value,
|
|
248
|
+
"span_count": len(spans),
|
|
249
|
+
"failed": failed,
|
|
250
|
+
},
|
|
251
|
+
},
|
|
252
|
+
source_type=SourceType.TRACE_MINING,
|
|
253
|
+
source_ref=trace_id,
|
|
254
|
+
created_at=now_ms(),
|
|
255
|
+
)
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
@dataclass
|
|
259
|
+
class MiningReport:
|
|
260
|
+
"""挖掘结果报告 — 条目 + skipped 明细 + 统计"""
|
|
261
|
+
|
|
262
|
+
strategy: str
|
|
263
|
+
candidates: int = 0
|
|
264
|
+
inspected: int = 0
|
|
265
|
+
items: list[EvalDatasetItem] = field(default_factory=list)
|
|
266
|
+
skipped: list[dict[str, Any]] = field(default_factory=list)
|
|
267
|
+
|
|
268
|
+
def to_dict(self) -> dict[str, Any]:
|
|
269
|
+
return {
|
|
270
|
+
"strategy": self.strategy,
|
|
271
|
+
"candidates": self.candidates,
|
|
272
|
+
"inspected": self.inspected,
|
|
273
|
+
"mined": len(self.items),
|
|
274
|
+
"skipped_count": len(self.skipped),
|
|
275
|
+
"skipped": self.skipped,
|
|
276
|
+
"item_ids": [i.id for i in self.items],
|
|
277
|
+
}
|