eval-builder 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- eval_builder/__init__.py +3 -0
- eval_builder/cli.py +354 -0
- eval_builder/data/SKILL.md +78 -0
- eval_builder/draft.py +271 -0
- eval_builder/export.py +284 -0
- eval_builder/ingest/__init__.py +181 -0
- eval_builder/ingest/formats.py +645 -0
- eval_builder/io.py +82 -0
- eval_builder/judge/__init__.py +1 -0
- eval_builder/judge/check.py +420 -0
- eval_builder/judge/plan.py +151 -0
- eval_builder/judge/run.py +149 -0
- eval_builder/judge/stats.py +76 -0
- eval_builder/mcp_server.py +175 -0
- eval_builder/redact.py +101 -0
- eval_builder/report.py +263 -0
- eval_builder/schema.py +242 -0
- eval_builder/select.py +412 -0
- eval_builder/setup_agents.py +190 -0
- eval_builder/status.py +32 -0
- eval_builder/workspace.py +67 -0
- eval_builder-0.1.0.dist-info/METADATA +154 -0
- eval_builder-0.1.0.dist-info/RECORD +26 -0
- eval_builder-0.1.0.dist-info/WHEEL +4 -0
- eval_builder-0.1.0.dist-info/entry_points.txt +2 -0
- eval_builder-0.1.0.dist-info/licenses/LICENSE +21 -0
eval_builder/report.py
ADDED
|
@@ -0,0 +1,263 @@
|
|
|
1
|
+
"""Markdown and JSON report over whatever steps have run in a workspace."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from datetime import UTC, datetime
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from . import __version__
|
|
10
|
+
from .draft import load_cases
|
|
11
|
+
from .io import read_json, write_json
|
|
12
|
+
from .workspace import Workspace
|
|
13
|
+
|
|
14
|
+
STATIC_LIMITS = [
|
|
15
|
+
"Selection is lexical (TF-IDF over the user input). Two inputs that mean the same thing "
|
|
16
|
+
"in different words can land in different clusters, and near-duplicate detection only "
|
|
17
|
+
"catches close textual matches.",
|
|
18
|
+
"Redaction is pattern-based (emails, common API key formats, Luhn-valid card numbers, "
|
|
19
|
+
"US-style phone numbers, SSN-shaped numbers). Names, addresses and free-form secrets are "
|
|
20
|
+
"not detected. Review traces before sharing them.",
|
|
21
|
+
"eval-builder does not write expected behavior. Cases are only as good as what the agent "
|
|
22
|
+
"and user put in cases.yaml.",
|
|
23
|
+
"Judge verdicts depend on the thresholds shown and on how many cases, trials and human "
|
|
24
|
+
"labels were provided. Small samples give wide intervals; read the intervals, not just "
|
|
25
|
+
"the point estimates.",
|
|
26
|
+
"The bias probes cover answer order and irrelevant padding only. They do not detect "
|
|
27
|
+
"self-preference, style bias or rubric misreadings.",
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
_DEDUPE_LABEL = {"input": "the user turns", "input+output": "the user turns plus the output"}
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _pct(x: float | None) -> str:
|
|
35
|
+
return "n/a" if x is None else f"{x:.0%}"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _ci(r: dict[str, Any] | None) -> str:
|
|
39
|
+
if not r or r.get("rate") is None:
|
|
40
|
+
return "n/a"
|
|
41
|
+
lo, hi = r["ci95"]
|
|
42
|
+
return f"{r['rate']:.0%} [{lo:.0%}, {hi:.0%}] (n={r['n']})"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _md_escape(s: Any) -> str:
|
|
46
|
+
return str(s).replace("|", "\\|").replace("\n", " ")
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def build_report(workspace: str | Path, title: str | None = None) -> dict[str, Any]:
|
|
50
|
+
ws = Workspace.at(workspace)
|
|
51
|
+
data: dict[str, Any] = {
|
|
52
|
+
"generated_at": datetime.now(UTC).isoformat(timespec="seconds"),
|
|
53
|
+
"eval_builder_version": __version__,
|
|
54
|
+
"workspace": str(ws.root),
|
|
55
|
+
}
|
|
56
|
+
if ws.ingest_report.exists():
|
|
57
|
+
data["ingest"] = read_json(ws.ingest_report)
|
|
58
|
+
if ws.selection.exists():
|
|
59
|
+
data["selection"] = read_json(ws.selection)
|
|
60
|
+
if ws.cases.exists():
|
|
61
|
+
cases = load_cases(ws.root).get("cases") or []
|
|
62
|
+
data["cases"] = {
|
|
63
|
+
"total": len(cases),
|
|
64
|
+
"by_status": {
|
|
65
|
+
s: sum(1 for c in cases if c.get("status") == s)
|
|
66
|
+
for s in ("draft", "ready", "dropped")
|
|
67
|
+
},
|
|
68
|
+
}
|
|
69
|
+
if ws.judge_check.exists():
|
|
70
|
+
data["judge_check"] = read_json(ws.judge_check)
|
|
71
|
+
manifest = ws.exports / "manifest.json"
|
|
72
|
+
if manifest.exists():
|
|
73
|
+
data["exports"] = read_json(manifest)
|
|
74
|
+
run_log = ws.root / "judge_run_log.json"
|
|
75
|
+
if run_log.exists():
|
|
76
|
+
data["judge_run"] = read_json(run_log)
|
|
77
|
+
data["limits"] = list(STATIC_LIMITS) + _dynamic_limits(data)
|
|
78
|
+
write_json(ws.report_json, data)
|
|
79
|
+
ws.report_md.write_text(render_markdown(data, title), encoding="utf-8")
|
|
80
|
+
return {
|
|
81
|
+
"report_md": str(ws.report_md),
|
|
82
|
+
"report_json": str(ws.report_json),
|
|
83
|
+
"sections": [
|
|
84
|
+
k for k in ("ingest", "selection", "cases", "judge_check", "exports") if k in data
|
|
85
|
+
],
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _dynamic_limits(data: dict[str, Any]) -> list[str]:
|
|
90
|
+
out = []
|
|
91
|
+
jc = data.get("judge_check")
|
|
92
|
+
if jc and not jc.get("labels_file"):
|
|
93
|
+
out.append("No human labels were provided, so no judge can be marked trustworthy.")
|
|
94
|
+
sel = data.get("selection")
|
|
95
|
+
if sel and sel["population"]["failures_unique"] == 0:
|
|
96
|
+
out.append(
|
|
97
|
+
"No traces carried an error flag or negative feedback, so failure "
|
|
98
|
+
"oversampling had nothing to work with."
|
|
99
|
+
)
|
|
100
|
+
return out
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def render_markdown(d: dict[str, Any], title: str | None = None) -> str:
|
|
104
|
+
L: list[str] = [f"# {title or 'eval-builder report'}", ""]
|
|
105
|
+
L.append(f"Generated {d['generated_at']} by eval-builder {d['eval_builder_version']}.")
|
|
106
|
+
L.append("")
|
|
107
|
+
ing = d.get("ingest")
|
|
108
|
+
if ing:
|
|
109
|
+
L += [
|
|
110
|
+
"## Sources",
|
|
111
|
+
"",
|
|
112
|
+
"| file | format | sha256 | records | traces | skipped |",
|
|
113
|
+
"|---|---|---|---|---|---|",
|
|
114
|
+
]
|
|
115
|
+
for s in ing["sources"]:
|
|
116
|
+
skipped = ", ".join(f"{k}: {v}" for k, v in s["skipped"].items()) or "0"
|
|
117
|
+
L.append(
|
|
118
|
+
f"| {_md_escape(Path(s['path']).name)} | {s['format']} | "
|
|
119
|
+
f"`{s['sha256'][:16]}` | {s['records']} | {s['traces']} | "
|
|
120
|
+
f"{_md_escape(skipped)} |"
|
|
121
|
+
)
|
|
122
|
+
L += [
|
|
123
|
+
"",
|
|
124
|
+
f"Traces ingested: {ing['traces']} ({ing['multi_turn']} multi-turn, "
|
|
125
|
+
f"{ing['with_error']} with an error flag). Feedback: "
|
|
126
|
+
+ ", ".join(f"{k} {v}" for k, v in ing["feedback"].items())
|
|
127
|
+
+ ".",
|
|
128
|
+
"",
|
|
129
|
+
]
|
|
130
|
+
r = ing["redactions"]
|
|
131
|
+
L += ["## Redactions", ""]
|
|
132
|
+
if not r["enabled"]:
|
|
133
|
+
L.append("Redaction was turned off for this run.")
|
|
134
|
+
elif r["total"] == 0:
|
|
135
|
+
L.append("Redaction was on. No matches.")
|
|
136
|
+
else:
|
|
137
|
+
L.append(
|
|
138
|
+
f"Redaction was on. {r['total']} value(s) replaced in "
|
|
139
|
+
f"{r['traces_affected']} trace(s): "
|
|
140
|
+
+ ", ".join(f"{k} {v}" for k, v in r["by_kind"].items())
|
|
141
|
+
+ "."
|
|
142
|
+
)
|
|
143
|
+
L.append("")
|
|
144
|
+
sel = d.get("selection")
|
|
145
|
+
if sel:
|
|
146
|
+
p = sel["population"]
|
|
147
|
+
L += [
|
|
148
|
+
"## Selection",
|
|
149
|
+
"",
|
|
150
|
+
f"{p['traces']} traces, {p['exact_duplicates_removed']} exact duplicates removed, "
|
|
151
|
+
f"{p['near_duplicates_merged']} near-duplicates merged (cosine >= "
|
|
152
|
+
f"{sel['params']['near_dup_threshold']} on "
|
|
153
|
+
f"{_DEDUPE_LABEL.get(sel['params']['dedupe_on'], sel['params']['dedupe_on'])}), "
|
|
154
|
+
f"{p['unique']} unique. Selected {sel['selected_count']} "
|
|
155
|
+
f"({sel['selected_failures']} failures; failures are {p['failures_unique']} of "
|
|
156
|
+
f"{p['unique']} unique traces). Seed {sel['params']['seed']}, "
|
|
157
|
+
f"{sel['params']['clusters']} k-means clusters.",
|
|
158
|
+
"",
|
|
159
|
+
"| cluster | traces | share | selected | top terms |",
|
|
160
|
+
"|---|---|---|---|---|",
|
|
161
|
+
]
|
|
162
|
+
for c in sel["clusters"]:
|
|
163
|
+
L.append(
|
|
164
|
+
f"| {c['id']} | {c['size']} | {c['share']:.0%} | {c['selected']} | "
|
|
165
|
+
f"{_md_escape(', '.join(c['terms']))} |"
|
|
166
|
+
)
|
|
167
|
+
if sel.get("strata"):
|
|
168
|
+
L += ["", "| stratum | value | unique traces | selected |", "|---|---|---|---|"]
|
|
169
|
+
for f, vals in sel["strata"].items():
|
|
170
|
+
for v, cnt in vals.items():
|
|
171
|
+
L.append(f"| {f} | {_md_escape(v)} | {cnt['population']} | {cnt['selected']} |")
|
|
172
|
+
if sel.get("strata_skipped"):
|
|
173
|
+
L.append("")
|
|
174
|
+
L.append(
|
|
175
|
+
"Not stratified: "
|
|
176
|
+
+ ", ".join(f"{k} ({v})" for k, v in sel["strata_skipped"].items())
|
|
177
|
+
)
|
|
178
|
+
L += ["", "| trace | cluster | represents | why it was picked |", "|---|---|---|---|"]
|
|
179
|
+
for s in sel["selected"]:
|
|
180
|
+
L.append(
|
|
181
|
+
f"| `{s['trace_id']}` | {s['cluster']} | {s['represents']} | "
|
|
182
|
+
f"{_md_escape('; '.join(s['reasons']))} |"
|
|
183
|
+
)
|
|
184
|
+
L.append("")
|
|
185
|
+
cs = d.get("cases")
|
|
186
|
+
if cs:
|
|
187
|
+
b = cs["by_status"]
|
|
188
|
+
L += [
|
|
189
|
+
"## Cases",
|
|
190
|
+
"",
|
|
191
|
+
f"{cs['total']} cases: {b['ready']} ready, {b['draft']} draft, {b['dropped']} dropped.",
|
|
192
|
+
"",
|
|
193
|
+
]
|
|
194
|
+
jc = d.get("judge_check")
|
|
195
|
+
if jc:
|
|
196
|
+
L += ["## Judge reliability", ""]
|
|
197
|
+
th = jc["thresholds"]
|
|
198
|
+
L.append(
|
|
199
|
+
f"Thresholds: flip rate <= {th['max_flip_rate']:.0%} (cases with >= "
|
|
200
|
+
f"{th['min_trials']} trials, >= {th['min_cases']} cases), position "
|
|
201
|
+
f"consistency >= {th['min_position_consistency']:.0%}, padding moves verdict "
|
|
202
|
+
f"toward padded answer <= {th['max_toward_padded_rate']:.0%}, Cohen's kappa "
|
|
203
|
+
f">= {th['min_kappa']} on >= {th['min_labeled']} human-labeled cases. Intervals "
|
|
204
|
+
f"are 95% (Wilson for rates, Cohen's large-sample SE for kappa). Self-agreement is "
|
|
205
|
+
"the chance one call matches the judge's own majority; majority-of-3 stable is "
|
|
206
|
+
"the chance a 3-call majority vote matches it (exact, needs >= 5 trials)."
|
|
207
|
+
)
|
|
208
|
+
L += [
|
|
209
|
+
"",
|
|
210
|
+
"| judge | mode | verdict | flip rate | self-agreement | majority-of-3 stable | "
|
|
211
|
+
"accuracy vs humans | kappa | position consistency | first-shown picked | "
|
|
212
|
+
"padding helped |",
|
|
213
|
+
"|---|---|---|---|---|---|---|---|---|---|---|",
|
|
214
|
+
]
|
|
215
|
+
for jid, r in jc["judges"].items():
|
|
216
|
+
st, ag = r["stability"], r["human_agreement"] or {}
|
|
217
|
+
pos, verb = r["position_probe"] or {}, r["verbosity_probe"] or {}
|
|
218
|
+
kappa = "n/a"
|
|
219
|
+
if ag.get("kappa") is not None:
|
|
220
|
+
kappa = f"{ag['kappa']:.2f}"
|
|
221
|
+
if ag.get("kappa_ci95"):
|
|
222
|
+
kappa += f" [{ag['kappa_ci95'][0]:.2f}, {ag['kappa_ci95'][1]:.2f}]"
|
|
223
|
+
L.append(
|
|
224
|
+
f"| {_md_escape(jid)} | {r['mode']} | **{r['verdict']}** | "
|
|
225
|
+
f"{_ci(st['flip_rate'])} | {_pct(st['mean_self_agreement'])} | "
|
|
226
|
+
f"{_pct(st.get('majority_of_3_stability'))} | "
|
|
227
|
+
f"{_ci(ag.get('accuracy'))} | {kappa} | {_ci(pos.get('consistency'))} | "
|
|
228
|
+
f"{_ci(pos.get('first_position_rate'))} | {_ci(verb.get('toward_padded'))} |"
|
|
229
|
+
)
|
|
230
|
+
L.append("")
|
|
231
|
+
for jid, r in jc["judges"].items():
|
|
232
|
+
L.append(
|
|
233
|
+
f"- **{jid}** ({r['verdict']}): "
|
|
234
|
+
+ " ".join(s.rstrip(".") + "." for s in r["reasons"])
|
|
235
|
+
)
|
|
236
|
+
L.append("")
|
|
237
|
+
run = d.get("judge_run")
|
|
238
|
+
if run:
|
|
239
|
+
L += ["Judge calls made through the opt-in judge plugin (`judge-run`):", ""]
|
|
240
|
+
for i, rr in enumerate(run.get("runs", [run]), 1):
|
|
241
|
+
for jid, e in rr["judges"].items():
|
|
242
|
+
if not e["requests"]:
|
|
243
|
+
continue
|
|
244
|
+
L.append(
|
|
245
|
+
f"- run {i}, {jid}: `{_md_escape(e['command'])}`, {e['ok']} ok, "
|
|
246
|
+
f"{e['errors']} errors, {e['seconds']} s"
|
|
247
|
+
)
|
|
248
|
+
L.append("")
|
|
249
|
+
ex = d.get("exports")
|
|
250
|
+
if ex:
|
|
251
|
+
L += [
|
|
252
|
+
"## Exports",
|
|
253
|
+
"",
|
|
254
|
+
f"{ex['cases']} ready cases exported.",
|
|
255
|
+
"",
|
|
256
|
+
"| file | sha256 |",
|
|
257
|
+
"|---|---|",
|
|
258
|
+
]
|
|
259
|
+
for f in ex["files"]:
|
|
260
|
+
L.append(f"| {f['path']} | `{f['sha256'][:16]}` |")
|
|
261
|
+
L.append("")
|
|
262
|
+
L += ["## Limits", ""] + [f"- {x}" for x in d["limits"]] + [""]
|
|
263
|
+
return "\n".join(L)
|
eval_builder/schema.py
ADDED
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
"""The normalized trace schema every ingest format maps into."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
import re
|
|
8
|
+
from dataclasses import asdict, dataclass, field
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass
|
|
13
|
+
class Trace:
|
|
14
|
+
"""One logged interaction, normalized.
|
|
15
|
+
|
|
16
|
+
`input` is the last user message before the final assistant reply and `output`
|
|
17
|
+
is that reply's text. `messages` keeps the whole conversation (including the
|
|
18
|
+
final reply) so multi-turn context is never lost.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
id: str
|
|
22
|
+
input: str
|
|
23
|
+
output: str
|
|
24
|
+
messages: list[dict[str, str]] = field(default_factory=list)
|
|
25
|
+
system: str | None = None
|
|
26
|
+
tools: list[str] = field(default_factory=list)
|
|
27
|
+
error: bool = False
|
|
28
|
+
feedback: str | None = None # "positive" | "negative" | "neutral" | None
|
|
29
|
+
route: str | None = None
|
|
30
|
+
model: str | None = None
|
|
31
|
+
metadata: dict[str, Any] = field(default_factory=dict)
|
|
32
|
+
source: dict[str, Any] = field(default_factory=dict)
|
|
33
|
+
|
|
34
|
+
def to_dict(self) -> dict[str, Any]:
|
|
35
|
+
return asdict(self)
|
|
36
|
+
|
|
37
|
+
@classmethod
|
|
38
|
+
def from_dict(cls, d: dict[str, Any]) -> Trace:
|
|
39
|
+
known = {f for f in cls.__dataclass_fields__}
|
|
40
|
+
return cls(**{k: v for k, v in d.items() if k in known})
|
|
41
|
+
|
|
42
|
+
def prior_turns(self) -> list[dict[str, str]]:
|
|
43
|
+
"""Conversation before the last user message (the case input).
|
|
44
|
+
|
|
45
|
+
Tool calls and intermediate assistant steps after that message are part of the
|
|
46
|
+
app's behavior being evaluated, so they are not included.
|
|
47
|
+
"""
|
|
48
|
+
last_user = max(
|
|
49
|
+
(i for i, m in enumerate(self.messages) if m.get("role") == "user"), default=0
|
|
50
|
+
)
|
|
51
|
+
return list(self.messages[:last_user])
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def text_of(content: Any) -> str:
|
|
55
|
+
"""Extract plain text from the many shapes providers use for message content."""
|
|
56
|
+
if content is None:
|
|
57
|
+
return ""
|
|
58
|
+
if isinstance(content, str):
|
|
59
|
+
return content
|
|
60
|
+
if isinstance(content, (int, float, bool)):
|
|
61
|
+
return str(content)
|
|
62
|
+
if isinstance(content, list):
|
|
63
|
+
parts = [text_of(p) for p in content]
|
|
64
|
+
return "\n".join(p for p in parts if p)
|
|
65
|
+
if isinstance(content, dict):
|
|
66
|
+
ptype = content.get("type")
|
|
67
|
+
if ptype in ("tool_use", "tool_call", "function_call", "image", "image_url", "input_image"):
|
|
68
|
+
return ""
|
|
69
|
+
for key in ("text", "content", "value", "output_text", "input_text"):
|
|
70
|
+
if key in content:
|
|
71
|
+
return text_of(content[key])
|
|
72
|
+
return ""
|
|
73
|
+
return str(content)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
_FEEDBACK_POS = {
|
|
77
|
+
"up",
|
|
78
|
+
"thumbs_up",
|
|
79
|
+
"thumbsup",
|
|
80
|
+
"positive",
|
|
81
|
+
"good",
|
|
82
|
+
"like",
|
|
83
|
+
"liked",
|
|
84
|
+
"yes",
|
|
85
|
+
"true",
|
|
86
|
+
"+1",
|
|
87
|
+
"1",
|
|
88
|
+
"helpful",
|
|
89
|
+
"pass",
|
|
90
|
+
"correct",
|
|
91
|
+
}
|
|
92
|
+
_FEEDBACK_NEG = {
|
|
93
|
+
"down",
|
|
94
|
+
"thumbs_down",
|
|
95
|
+
"thumbsdown",
|
|
96
|
+
"negative",
|
|
97
|
+
"bad",
|
|
98
|
+
"dislike",
|
|
99
|
+
"disliked",
|
|
100
|
+
"no",
|
|
101
|
+
"false",
|
|
102
|
+
"-1",
|
|
103
|
+
"0",
|
|
104
|
+
"unhelpful",
|
|
105
|
+
"fail",
|
|
106
|
+
"incorrect",
|
|
107
|
+
"wrong",
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def normalize_feedback(value: Any) -> str | None:
|
|
112
|
+
"""Map thumbs, booleans, 0/1 and 1-5 ratings onto positive/negative/neutral."""
|
|
113
|
+
if value is None:
|
|
114
|
+
return None
|
|
115
|
+
if isinstance(value, bool):
|
|
116
|
+
return "positive" if value else "negative"
|
|
117
|
+
if isinstance(value, (int, float)):
|
|
118
|
+
v = float(value)
|
|
119
|
+
if v < 0:
|
|
120
|
+
return "negative"
|
|
121
|
+
if v <= 1:
|
|
122
|
+
return "negative" if v < 0.5 else "positive"
|
|
123
|
+
if v <= 5:
|
|
124
|
+
if v <= 2:
|
|
125
|
+
return "negative"
|
|
126
|
+
return "positive" if v >= 4 else "neutral"
|
|
127
|
+
return None
|
|
128
|
+
if isinstance(value, str):
|
|
129
|
+
s = value.strip().lower().replace(" ", "_")
|
|
130
|
+
if s in _FEEDBACK_POS:
|
|
131
|
+
return "positive"
|
|
132
|
+
if s in _FEEDBACK_NEG:
|
|
133
|
+
return "negative"
|
|
134
|
+
if s in ("neutral", "meh", "mixed"):
|
|
135
|
+
return "neutral"
|
|
136
|
+
try:
|
|
137
|
+
return normalize_feedback(float(s))
|
|
138
|
+
except ValueError:
|
|
139
|
+
return None
|
|
140
|
+
if isinstance(value, dict):
|
|
141
|
+
for key in ("value", "score", "rating", "label"):
|
|
142
|
+
if key in value:
|
|
143
|
+
return normalize_feedback(value[key])
|
|
144
|
+
return None
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
FEEDBACK_KEYS = ("feedback", "user_feedback", "thumbs", "rating", "user_rating", "vote")
|
|
148
|
+
ERROR_KEYS = ("error", "is_error", "failed", "exception")
|
|
149
|
+
ROUTE_KEYS = ("route", "endpoint", "path", "feature", "workflow", "task", "intent")
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def fill_common_fields(trace: Trace) -> Trace:
|
|
153
|
+
"""Derive error, feedback and route from metadata when a format did not set them."""
|
|
154
|
+
md = trace.metadata
|
|
155
|
+
if trace.feedback is None:
|
|
156
|
+
for key in FEEDBACK_KEYS:
|
|
157
|
+
if key in md:
|
|
158
|
+
trace.feedback = normalize_feedback(md[key])
|
|
159
|
+
if trace.feedback:
|
|
160
|
+
break
|
|
161
|
+
if not trace.error:
|
|
162
|
+
for key in ERROR_KEYS:
|
|
163
|
+
v = md.get(key)
|
|
164
|
+
if v not in (None, False, "", 0, "false", "False", "none", "None"):
|
|
165
|
+
trace.error = True
|
|
166
|
+
break
|
|
167
|
+
status = str(md.get("status", "")).lower()
|
|
168
|
+
if status in ("error", "failed", "failure"):
|
|
169
|
+
trace.error = True
|
|
170
|
+
if trace.route is None:
|
|
171
|
+
for key in ROUTE_KEYS:
|
|
172
|
+
v = md.get(key)
|
|
173
|
+
if isinstance(v, (str, int)) and str(v):
|
|
174
|
+
trace.route = str(v)
|
|
175
|
+
break
|
|
176
|
+
if trace.model is None and isinstance(md.get("model"), str):
|
|
177
|
+
trace.model = md["model"]
|
|
178
|
+
return trace
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
_ID_SAFE = re.compile(r"[^A-Za-z0-9_.:-]+")
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def make_id(*parts: Any, given: Any = None) -> str:
|
|
185
|
+
if given not in (None, ""):
|
|
186
|
+
cleaned = _ID_SAFE.sub("-", str(given)).strip("-")
|
|
187
|
+
if cleaned:
|
|
188
|
+
return cleaned[:80]
|
|
189
|
+
blob = json.dumps(parts, ensure_ascii=False, sort_keys=True, default=str)
|
|
190
|
+
return "t-" + hashlib.sha1(blob.encode("utf-8")).hexdigest()[:12]
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def build_trace(
|
|
194
|
+
messages: list[dict[str, Any]],
|
|
195
|
+
*,
|
|
196
|
+
fmt: str,
|
|
197
|
+
file: str,
|
|
198
|
+
index: int,
|
|
199
|
+
given_id: Any = None,
|
|
200
|
+
system: str | None = None,
|
|
201
|
+
tools: list[str] | None = None,
|
|
202
|
+
metadata: dict[str, Any] | None = None,
|
|
203
|
+
error: bool = False,
|
|
204
|
+
feedback: str | None = None,
|
|
205
|
+
route: str | None = None,
|
|
206
|
+
model: str | None = None,
|
|
207
|
+
) -> Trace:
|
|
208
|
+
"""Build a Trace from role/content messages (content already flattened to text)."""
|
|
209
|
+
norm: list[dict[str, str]] = []
|
|
210
|
+
sys_parts: list[str] = [system] if system else []
|
|
211
|
+
for m in messages:
|
|
212
|
+
role = str(m.get("role", "user")).lower()
|
|
213
|
+
if role in ("system", "developer"):
|
|
214
|
+
t = text_of(m.get("content"))
|
|
215
|
+
if t:
|
|
216
|
+
sys_parts.append(t)
|
|
217
|
+
continue
|
|
218
|
+
if role in ("human",):
|
|
219
|
+
role = "user"
|
|
220
|
+
if role in ("ai", "model", "bot"):
|
|
221
|
+
role = "assistant"
|
|
222
|
+
norm.append({"role": role, "content": text_of(m.get("content"))})
|
|
223
|
+
output = ""
|
|
224
|
+
if norm and norm[-1]["role"] == "assistant":
|
|
225
|
+
output = norm[-1]["content"]
|
|
226
|
+
user_msgs = [m["content"] for m in norm if m["role"] == "user"]
|
|
227
|
+
inp = user_msgs[-1] if user_msgs else ""
|
|
228
|
+
trace = Trace(
|
|
229
|
+
id=make_id(fmt, file, index, inp, output, given=given_id),
|
|
230
|
+
input=inp,
|
|
231
|
+
output=output,
|
|
232
|
+
messages=norm,
|
|
233
|
+
system="\n\n".join(sys_parts) or None,
|
|
234
|
+
tools=sorted(set(tools or [])),
|
|
235
|
+
error=error,
|
|
236
|
+
feedback=feedback,
|
|
237
|
+
route=route,
|
|
238
|
+
model=model,
|
|
239
|
+
metadata=dict(metadata or {}),
|
|
240
|
+
source={"format": fmt, "file": file, "index": index},
|
|
241
|
+
)
|
|
242
|
+
return fill_common_fields(trace)
|