eval-builder 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- eval_builder/__init__.py +3 -0
- eval_builder/cli.py +354 -0
- eval_builder/data/SKILL.md +78 -0
- eval_builder/draft.py +271 -0
- eval_builder/export.py +284 -0
- eval_builder/ingest/__init__.py +181 -0
- eval_builder/ingest/formats.py +645 -0
- eval_builder/io.py +82 -0
- eval_builder/judge/__init__.py +1 -0
- eval_builder/judge/check.py +420 -0
- eval_builder/judge/plan.py +151 -0
- eval_builder/judge/run.py +149 -0
- eval_builder/judge/stats.py +76 -0
- eval_builder/mcp_server.py +175 -0
- eval_builder/redact.py +101 -0
- eval_builder/report.py +263 -0
- eval_builder/schema.py +242 -0
- eval_builder/select.py +412 -0
- eval_builder/setup_agents.py +190 -0
- eval_builder/status.py +32 -0
- eval_builder/workspace.py +67 -0
- eval_builder-0.1.0.dist-info/METADATA +154 -0
- eval_builder-0.1.0.dist-info/RECORD +26 -0
- eval_builder-0.1.0.dist-info/WHEEL +4 -0
- eval_builder-0.1.0.dist-info/entry_points.txt +2 -0
- eval_builder-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,645 @@
|
|
|
1
|
+
"""Parsers for each supported log format. Each yields normalized Trace objects."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from collections.abc import Iterator
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from ..schema import FEEDBACK_KEYS, Trace, build_trace, normalize_feedback, text_of
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class SkipRecord(Exception):
|
|
13
|
+
"""Raised when a record cannot become a trace; the message is the reason."""
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
FORMATS = ("openai", "anthropic", "langfuse", "otel", "generic")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
# --------------------------------------------------------------------------- detection
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def detect_format(records: list[Any]) -> str:
|
|
23
|
+
"""Guess the format from the first few records. Raises ValueError if unsure."""
|
|
24
|
+
votes: dict[str, int] = {}
|
|
25
|
+
for rec in records[:20]:
|
|
26
|
+
fmt = _detect_one(rec)
|
|
27
|
+
if fmt:
|
|
28
|
+
votes[fmt] = votes.get(fmt, 0) + 1
|
|
29
|
+
if not votes:
|
|
30
|
+
raise ValueError(
|
|
31
|
+
"could not detect the log format; pass --format "
|
|
32
|
+
"(openai, anthropic, langfuse, otel, generic)"
|
|
33
|
+
)
|
|
34
|
+
return max(votes.items(), key=lambda kv: kv[1])[0]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _detect_one(rec: Any) -> str | None:
|
|
38
|
+
if not isinstance(rec, dict):
|
|
39
|
+
return None
|
|
40
|
+
if (
|
|
41
|
+
"resourceSpans" in rec
|
|
42
|
+
or "spanId" in rec
|
|
43
|
+
or "span_id" in rec
|
|
44
|
+
or (isinstance(rec.get("context"), dict) and "span_id" in rec["context"])
|
|
45
|
+
):
|
|
46
|
+
return "otel"
|
|
47
|
+
if "observations" in rec or "htmlPath" in rec or "projectId" in rec:
|
|
48
|
+
return "langfuse"
|
|
49
|
+
if _looks_anthropic(rec):
|
|
50
|
+
return "anthropic"
|
|
51
|
+
if (
|
|
52
|
+
"messages" in rec
|
|
53
|
+
or "choices" in rec
|
|
54
|
+
or (isinstance(rec.get("request"), dict) and "messages" in rec["request"])
|
|
55
|
+
):
|
|
56
|
+
return "openai"
|
|
57
|
+
if "input" in rec and "output" in rec and isinstance(rec.get("scores"), list):
|
|
58
|
+
return "langfuse"
|
|
59
|
+
if any(k in rec for k in ("input", "prompt", "question")) and any(
|
|
60
|
+
k in rec for k in ("output", "completion", "answer", "response")
|
|
61
|
+
):
|
|
62
|
+
return "generic"
|
|
63
|
+
return None
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _looks_anthropic(rec: dict[str, Any]) -> bool:
|
|
67
|
+
req = _sub(rec, "request")
|
|
68
|
+
resp = rec.get("response")
|
|
69
|
+
if (
|
|
70
|
+
isinstance(resp, dict)
|
|
71
|
+
and isinstance(resp.get("content"), list)
|
|
72
|
+
and ("stop_reason" in resp or resp.get("type") == "message")
|
|
73
|
+
):
|
|
74
|
+
return True
|
|
75
|
+
if rec.get("type") == "message" and isinstance(rec.get("content"), list):
|
|
76
|
+
return True
|
|
77
|
+
if isinstance(req, dict) and "messages" in req:
|
|
78
|
+
if isinstance(req.get("system"), (str, list)) and "choices" not in rec:
|
|
79
|
+
return True
|
|
80
|
+
for m in req.get("messages") or []:
|
|
81
|
+
content = m.get("content") if isinstance(m, dict) else None
|
|
82
|
+
if isinstance(content, list) and any(
|
|
83
|
+
isinstance(b, dict) and b.get("type") in ("tool_use", "tool_result")
|
|
84
|
+
for b in content
|
|
85
|
+
):
|
|
86
|
+
return True
|
|
87
|
+
return False
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
# --------------------------------------------------------------------------- helpers
|
|
91
|
+
|
|
92
|
+
_RESERVED_OPENAI = {
|
|
93
|
+
"messages",
|
|
94
|
+
"request",
|
|
95
|
+
"response",
|
|
96
|
+
"choices",
|
|
97
|
+
"metadata",
|
|
98
|
+
"id",
|
|
99
|
+
"model",
|
|
100
|
+
"usage",
|
|
101
|
+
"object",
|
|
102
|
+
"created",
|
|
103
|
+
"system_fingerprint",
|
|
104
|
+
"tools",
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _scalar_extras(rec: dict[str, Any], reserved: set[str]) -> dict[str, Any]:
|
|
109
|
+
out: dict[str, Any] = {}
|
|
110
|
+
for k, v in rec.items():
|
|
111
|
+
if k in reserved:
|
|
112
|
+
continue
|
|
113
|
+
if (
|
|
114
|
+
isinstance(v, (str, int, float, bool))
|
|
115
|
+
or v is None
|
|
116
|
+
or isinstance(v, list)
|
|
117
|
+
and all(isinstance(x, (str, int, float)) for x in v)
|
|
118
|
+
):
|
|
119
|
+
out[k] = v
|
|
120
|
+
return out
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _sub(rec: dict[str, Any], key: str) -> dict[str, Any]:
|
|
124
|
+
"""rec[key] when it is a dict, else rec itself (flat records)."""
|
|
125
|
+
v = rec.get(key)
|
|
126
|
+
return v if isinstance(v, dict) else rec
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _dict_or_empty(v: Any) -> dict[str, Any]:
|
|
130
|
+
return v if isinstance(v, dict) else {}
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _openai_tool_names(msg: dict[str, Any]) -> list[str]:
|
|
134
|
+
names = []
|
|
135
|
+
for tc in msg.get("tool_calls") or []:
|
|
136
|
+
if isinstance(tc, dict):
|
|
137
|
+
fn = tc.get("function") or {}
|
|
138
|
+
name = fn.get("name") or tc.get("name")
|
|
139
|
+
if name:
|
|
140
|
+
names.append(str(name))
|
|
141
|
+
fc = msg.get("function_call")
|
|
142
|
+
if isinstance(fc, dict) and fc.get("name"):
|
|
143
|
+
names.append(str(fc["name"]))
|
|
144
|
+
return names
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
# --------------------------------------------------------------------------- openai
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def parse_openai(rec: dict[str, Any], file: str, index: int) -> Trace:
|
|
151
|
+
req = _sub(rec, "request")
|
|
152
|
+
messages = list(req.get("messages") or [])
|
|
153
|
+
resp = _sub(rec, "response")
|
|
154
|
+
choices = resp.get("choices") if isinstance(resp, dict) else None
|
|
155
|
+
if isinstance(choices, list) and choices:
|
|
156
|
+
msg = choices[0].get("message") or {"role": "assistant", "content": choices[0].get("text")}
|
|
157
|
+
messages.append(msg)
|
|
158
|
+
if not messages:
|
|
159
|
+
raise SkipRecord("no messages")
|
|
160
|
+
tools: list[str] = []
|
|
161
|
+
for m in messages:
|
|
162
|
+
if isinstance(m, dict):
|
|
163
|
+
tools.extend(_openai_tool_names(m))
|
|
164
|
+
md = dict(rec.get("metadata") or {}) if isinstance(rec.get("metadata"), dict) else {}
|
|
165
|
+
md.update(_scalar_extras(rec, _RESERVED_OPENAI))
|
|
166
|
+
model = rec.get("model") or req.get("model") or (resp.get("model") if resp else None)
|
|
167
|
+
err = bool(rec.get("error")) or bool(isinstance(resp, dict) and resp.get("error"))
|
|
168
|
+
return build_trace(
|
|
169
|
+
[m for m in messages if isinstance(m, dict)],
|
|
170
|
+
fmt="openai",
|
|
171
|
+
file=file,
|
|
172
|
+
index=index,
|
|
173
|
+
given_id=rec.get("id"),
|
|
174
|
+
tools=tools,
|
|
175
|
+
metadata=md,
|
|
176
|
+
model=model if isinstance(model, str) else None,
|
|
177
|
+
error=err,
|
|
178
|
+
)
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
# --------------------------------------------------------------------------- anthropic
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _anthropic_message(m: dict[str, Any], tools: list[str]) -> dict[str, Any]:
|
|
185
|
+
content = m.get("content")
|
|
186
|
+
role = m.get("role", "user")
|
|
187
|
+
if isinstance(content, list):
|
|
188
|
+
texts = []
|
|
189
|
+
only_tool_results = bool(content)
|
|
190
|
+
for b in content:
|
|
191
|
+
if not isinstance(b, dict):
|
|
192
|
+
texts.append(str(b))
|
|
193
|
+
only_tool_results = False
|
|
194
|
+
continue
|
|
195
|
+
btype = b.get("type")
|
|
196
|
+
if btype == "tool_use":
|
|
197
|
+
if b.get("name"):
|
|
198
|
+
tools.append(str(b["name"]))
|
|
199
|
+
continue
|
|
200
|
+
if btype == "tool_result":
|
|
201
|
+
texts.append(text_of(b.get("content")))
|
|
202
|
+
continue
|
|
203
|
+
only_tool_results = False
|
|
204
|
+
if btype in ("text", None):
|
|
205
|
+
texts.append(text_of(b))
|
|
206
|
+
if role == "user" and only_tool_results:
|
|
207
|
+
role = "tool"
|
|
208
|
+
return {"role": role, "content": "\n".join(t for t in texts if t)}
|
|
209
|
+
return {"role": role, "content": text_of(content)}
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def parse_anthropic(rec: dict[str, Any], file: str, index: int) -> Trace:
|
|
213
|
+
req = _sub(rec, "request")
|
|
214
|
+
tools: list[str] = []
|
|
215
|
+
messages = [
|
|
216
|
+
_anthropic_message(m, tools) for m in req.get("messages") or [] if isinstance(m, dict)
|
|
217
|
+
]
|
|
218
|
+
resp = rec.get("response") if isinstance(rec.get("response"), dict) else None
|
|
219
|
+
if resp is None and rec.get("type") == "message" and "content" in rec and rec is not req:
|
|
220
|
+
resp = rec
|
|
221
|
+
if resp is not None and isinstance(resp.get("content"), list):
|
|
222
|
+
messages.append(
|
|
223
|
+
_anthropic_message({"role": "assistant", "content": resp["content"]}, tools)
|
|
224
|
+
)
|
|
225
|
+
if not messages:
|
|
226
|
+
raise SkipRecord("no messages")
|
|
227
|
+
system = req.get("system")
|
|
228
|
+
md = dict(rec.get("metadata") or {}) if isinstance(rec.get("metadata"), dict) else {}
|
|
229
|
+
md.update(_scalar_extras(rec, _RESERVED_OPENAI | {"system", "type", "content", "stop_reason"}))
|
|
230
|
+
if resp is not None and resp.get("stop_reason"):
|
|
231
|
+
md["stop_reason"] = resp["stop_reason"]
|
|
232
|
+
model = req.get("model") or (resp or {}).get("model")
|
|
233
|
+
err = bool(rec.get("error")) or (resp is not None and resp.get("type") == "error")
|
|
234
|
+
return build_trace(
|
|
235
|
+
messages,
|
|
236
|
+
fmt="anthropic",
|
|
237
|
+
file=file,
|
|
238
|
+
index=index,
|
|
239
|
+
given_id=rec.get("id") or (resp or {}).get("id"),
|
|
240
|
+
system=text_of(system) if system else None,
|
|
241
|
+
tools=tools,
|
|
242
|
+
metadata=md,
|
|
243
|
+
model=model if isinstance(model, str) else None,
|
|
244
|
+
error=err,
|
|
245
|
+
)
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
# --------------------------------------------------------------------------- langfuse
|
|
249
|
+
|
|
250
|
+
_TEXT_KEYS = (
|
|
251
|
+
"input",
|
|
252
|
+
"query",
|
|
253
|
+
"question",
|
|
254
|
+
"prompt",
|
|
255
|
+
"text",
|
|
256
|
+
"content",
|
|
257
|
+
"message",
|
|
258
|
+
"output",
|
|
259
|
+
"answer",
|
|
260
|
+
"completion",
|
|
261
|
+
"response",
|
|
262
|
+
"result",
|
|
263
|
+
)
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _lf_messages(value: Any, default_role: str) -> list[dict[str, Any]]:
|
|
267
|
+
if value is None:
|
|
268
|
+
return []
|
|
269
|
+
if (
|
|
270
|
+
isinstance(value, list)
|
|
271
|
+
and value
|
|
272
|
+
and all(isinstance(m, dict) and "role" in m for m in value)
|
|
273
|
+
):
|
|
274
|
+
return value
|
|
275
|
+
if isinstance(value, dict):
|
|
276
|
+
if isinstance(value.get("messages"), list):
|
|
277
|
+
return _lf_messages(value["messages"], default_role)
|
|
278
|
+
if "role" in value and "content" in value:
|
|
279
|
+
return [value]
|
|
280
|
+
if isinstance(value.get("choices"), list) and value["choices"]:
|
|
281
|
+
msg = value["choices"][0].get("message")
|
|
282
|
+
if isinstance(msg, dict):
|
|
283
|
+
return [msg]
|
|
284
|
+
for key in _TEXT_KEYS:
|
|
285
|
+
if isinstance(value.get(key), str):
|
|
286
|
+
return [{"role": default_role, "content": value[key]}]
|
|
287
|
+
return [{"role": default_role, "content": json.dumps(value, ensure_ascii=False)}]
|
|
288
|
+
if isinstance(value, str):
|
|
289
|
+
return [{"role": default_role, "content": value}]
|
|
290
|
+
return [{"role": default_role, "content": text_of(value)}]
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
def parse_langfuse(rec: dict[str, Any], file: str, index: int) -> Trace:
|
|
294
|
+
observations = [o for o in rec.get("observations") or [] if isinstance(o, dict)]
|
|
295
|
+
generations = [o for o in observations if o.get("type") == "GENERATION"]
|
|
296
|
+
generations.sort(key=lambda o: str(o.get("startTime") or ""))
|
|
297
|
+
inp, out = rec.get("input"), rec.get("output")
|
|
298
|
+
model = None
|
|
299
|
+
tools: list[str] = []
|
|
300
|
+
if generations:
|
|
301
|
+
last = generations[-1]
|
|
302
|
+
model = last.get("model")
|
|
303
|
+
if inp is None:
|
|
304
|
+
inp = last.get("input")
|
|
305
|
+
if out is None:
|
|
306
|
+
out = last.get("output")
|
|
307
|
+
for o in observations:
|
|
308
|
+
if o.get("type") == "TOOL" and o.get("name"):
|
|
309
|
+
tools.append(str(o["name"]))
|
|
310
|
+
if o.get("type") == "GENERATION":
|
|
311
|
+
for m in _lf_messages(o.get("output"), "assistant"):
|
|
312
|
+
tools.extend(_openai_tool_names(m))
|
|
313
|
+
messages = _lf_messages(inp, "user")
|
|
314
|
+
out_msgs = _lf_messages(out, "assistant")
|
|
315
|
+
if out_msgs:
|
|
316
|
+
final = out_msgs[-1]
|
|
317
|
+
messages = messages + [{"role": "assistant", "content": final.get("content")}]
|
|
318
|
+
tools.extend(_openai_tool_names(final))
|
|
319
|
+
if not messages:
|
|
320
|
+
raise SkipRecord("no input or output")
|
|
321
|
+
error = any(str(o.get("level", "")).upper() == "ERROR" for o in observations)
|
|
322
|
+
if str(rec.get("level", "")).upper() == "ERROR":
|
|
323
|
+
error = True
|
|
324
|
+
feedback = None
|
|
325
|
+
other_scores: dict[str, Any] = {}
|
|
326
|
+
for s in rec.get("scores") or []:
|
|
327
|
+
if not isinstance(s, dict):
|
|
328
|
+
continue
|
|
329
|
+
name = str(s.get("name", "")).lower().replace("-", "_").replace(" ", "_")
|
|
330
|
+
value = s.get("value", s.get("stringValue"))
|
|
331
|
+
if name in FEEDBACK_KEYS or "feedback" in name or "thumb" in name:
|
|
332
|
+
feedback = feedback or normalize_feedback(value)
|
|
333
|
+
else:
|
|
334
|
+
other_scores[s.get("name", "score")] = value
|
|
335
|
+
md = dict(rec.get("metadata") or {}) if isinstance(rec.get("metadata"), dict) else {}
|
|
336
|
+
if rec.get("tags"):
|
|
337
|
+
md["tags"] = rec["tags"]
|
|
338
|
+
for key in ("release", "version", "environment"):
|
|
339
|
+
if rec.get(key):
|
|
340
|
+
md[key] = rec[key]
|
|
341
|
+
if other_scores:
|
|
342
|
+
md["scores"] = other_scores
|
|
343
|
+
return build_trace(
|
|
344
|
+
messages,
|
|
345
|
+
fmt="langfuse",
|
|
346
|
+
file=file,
|
|
347
|
+
index=index,
|
|
348
|
+
given_id=rec.get("id"),
|
|
349
|
+
tools=tools,
|
|
350
|
+
metadata=md,
|
|
351
|
+
error=error,
|
|
352
|
+
feedback=feedback,
|
|
353
|
+
route=str(rec["name"]) if rec.get("name") else None,
|
|
354
|
+
model=model if isinstance(model, str) else None,
|
|
355
|
+
)
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
# --------------------------------------------------------------------------- otel
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
def _otel_value(v: Any) -> Any:
|
|
362
|
+
if not isinstance(v, dict):
|
|
363
|
+
return v
|
|
364
|
+
for key in ("stringValue", "boolValue", "doubleValue"):
|
|
365
|
+
if key in v:
|
|
366
|
+
return v[key]
|
|
367
|
+
if "intValue" in v:
|
|
368
|
+
try:
|
|
369
|
+
return int(v["intValue"])
|
|
370
|
+
except (TypeError, ValueError):
|
|
371
|
+
return v["intValue"]
|
|
372
|
+
if "arrayValue" in v:
|
|
373
|
+
return [_otel_value(x) for x in (v["arrayValue"] or {}).get("values", [])]
|
|
374
|
+
if "kvlistValue" in v:
|
|
375
|
+
return {
|
|
376
|
+
kv["key"]: _otel_value(kv.get("value")) for kv in v["kvlistValue"].get("values", [])
|
|
377
|
+
}
|
|
378
|
+
return v
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
def _otel_attrs(attrs: Any) -> dict[str, Any]:
|
|
382
|
+
if isinstance(attrs, dict):
|
|
383
|
+
return dict(attrs)
|
|
384
|
+
out: dict[str, Any] = {}
|
|
385
|
+
for kv in attrs or []:
|
|
386
|
+
if isinstance(kv, dict) and "key" in kv:
|
|
387
|
+
out[kv["key"]] = _otel_value(kv.get("value"))
|
|
388
|
+
return out
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def iter_otel_spans(rec: dict[str, Any]) -> Iterator[dict[str, Any]]:
|
|
392
|
+
"""Flatten OTLP JSON (resourceSpans) or pass through a single flat span."""
|
|
393
|
+
if "resourceSpans" in rec:
|
|
394
|
+
for rs in rec.get("resourceSpans") or []:
|
|
395
|
+
res_attrs = _otel_attrs((rs.get("resource") or {}).get("attributes"))
|
|
396
|
+
for ss in rs.get("scopeSpans") or rs.get("instrumentationLibrarySpans") or []:
|
|
397
|
+
for span in ss.get("spans") or []:
|
|
398
|
+
yield {**span, "_resource": res_attrs}
|
|
399
|
+
else:
|
|
400
|
+
yield rec
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
def _span_trace_id(span: dict[str, Any]) -> str:
|
|
404
|
+
ctx = _dict_or_empty(span.get("context"))
|
|
405
|
+
return str(span.get("traceId") or span.get("trace_id") or ctx.get("trace_id") or "")
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
def _span_is_error(span: dict[str, Any]) -> bool:
|
|
409
|
+
status = span.get("status") or {}
|
|
410
|
+
code = status.get("code", status.get("status_code"))
|
|
411
|
+
return code in (2, "2", "STATUS_CODE_ERROR", "ERROR") or bool(
|
|
412
|
+
_otel_attrs(span.get("attributes")).get("error.type")
|
|
413
|
+
)
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
def _parse_json_attr(v: Any) -> Any:
|
|
417
|
+
if isinstance(v, str):
|
|
418
|
+
try:
|
|
419
|
+
return json.loads(v)
|
|
420
|
+
except ValueError:
|
|
421
|
+
return v
|
|
422
|
+
return v
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
def _semconv_messages(raw: Any, tools: list[str]) -> list[dict[str, Any]]:
|
|
426
|
+
"""gen_ai.input.messages / gen_ai.output.messages: [{role, parts:[{type, content}]}]."""
|
|
427
|
+
raw = _parse_json_attr(raw)
|
|
428
|
+
msgs: list[dict[str, Any]] = []
|
|
429
|
+
if not isinstance(raw, list):
|
|
430
|
+
return msgs
|
|
431
|
+
for m in raw:
|
|
432
|
+
if not isinstance(m, dict):
|
|
433
|
+
continue
|
|
434
|
+
parts = m.get("parts")
|
|
435
|
+
if isinstance(parts, list):
|
|
436
|
+
texts = []
|
|
437
|
+
for p in parts:
|
|
438
|
+
if not isinstance(p, dict):
|
|
439
|
+
continue
|
|
440
|
+
if p.get("type") == "tool_call":
|
|
441
|
+
if p.get("name"):
|
|
442
|
+
tools.append(str(p["name"]))
|
|
443
|
+
continue
|
|
444
|
+
if p.get("type") in ("text", "tool_call_response", None):
|
|
445
|
+
texts.append(text_of(p.get("content", p.get("response", p.get("result")))))
|
|
446
|
+
content = "\n".join(t for t in texts if t)
|
|
447
|
+
else:
|
|
448
|
+
content = text_of(m.get("content"))
|
|
449
|
+
role = m.get("role", "user")
|
|
450
|
+
msgs.append({"role": "tool" if role == "tool" else role, "content": content})
|
|
451
|
+
return msgs
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
def _indexed_messages(attrs: dict[str, Any], prefix: str) -> list[dict[str, Any]]:
|
|
455
|
+
"""OpenLLMetry style gen_ai.prompt.0.role / gen_ai.prompt.0.content."""
|
|
456
|
+
idx: dict[int, dict[str, Any]] = {}
|
|
457
|
+
for k, v in attrs.items():
|
|
458
|
+
if not k.startswith(prefix + "."):
|
|
459
|
+
continue
|
|
460
|
+
rest = k[len(prefix) + 1 :].split(".", 1)
|
|
461
|
+
if len(rest) == 2 and rest[0].isdigit() and rest[1] in ("role", "content"):
|
|
462
|
+
idx.setdefault(int(rest[0]), {})[rest[1]] = v
|
|
463
|
+
return [idx[i] for i in sorted(idx)]
|
|
464
|
+
|
|
465
|
+
|
|
466
|
+
def _event_messages(span: dict[str, Any]) -> list[dict[str, Any]]:
|
|
467
|
+
msgs = []
|
|
468
|
+
for ev in span.get("events") or []:
|
|
469
|
+
name = ev.get("name", "")
|
|
470
|
+
attrs = _otel_attrs(ev.get("attributes"))
|
|
471
|
+
body = _parse_json_attr(ev.get("body", attrs.get("body")))
|
|
472
|
+
payload = body if isinstance(body, dict) else attrs
|
|
473
|
+
if name in (
|
|
474
|
+
"gen_ai.system.message",
|
|
475
|
+
"gen_ai.user.message",
|
|
476
|
+
"gen_ai.assistant.message",
|
|
477
|
+
"gen_ai.tool.message",
|
|
478
|
+
):
|
|
479
|
+
role = name.split(".")[1]
|
|
480
|
+
msgs.append({"role": role, "content": text_of(payload.get("content"))})
|
|
481
|
+
elif name == "gen_ai.choice":
|
|
482
|
+
message = _dict_or_empty(payload.get("message")) or payload
|
|
483
|
+
msgs.append({"role": "assistant", "content": text_of(message.get("content"))})
|
|
484
|
+
return msgs
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
def _span_messages(
|
|
488
|
+
span: dict[str, Any], tools: list[str]
|
|
489
|
+
) -> tuple[list[dict[str, Any]], str | None]:
|
|
490
|
+
attrs = _otel_attrs(span.get("attributes"))
|
|
491
|
+
system = attrs.get("gen_ai.system_instructions")
|
|
492
|
+
system_text = None
|
|
493
|
+
if system is not None:
|
|
494
|
+
parsed = _parse_json_attr(system)
|
|
495
|
+
system_text = text_of(parsed) or (system if isinstance(system, str) else None)
|
|
496
|
+
if "gen_ai.input.messages" in attrs or "gen_ai.output.messages" in attrs:
|
|
497
|
+
msgs = _semconv_messages(attrs.get("gen_ai.input.messages"), tools)
|
|
498
|
+
msgs += _semconv_messages(attrs.get("gen_ai.output.messages"), tools)
|
|
499
|
+
return msgs, system_text
|
|
500
|
+
prompt = _indexed_messages(attrs, "gen_ai.prompt")
|
|
501
|
+
completion = _indexed_messages(attrs, "gen_ai.completion")
|
|
502
|
+
if prompt or completion:
|
|
503
|
+
return prompt + [
|
|
504
|
+
{"role": m.get("role", "assistant"), "content": m.get("content")} for m in completion
|
|
505
|
+
], system_text
|
|
506
|
+
if isinstance(attrs.get("gen_ai.prompt"), str) or isinstance(
|
|
507
|
+
attrs.get("gen_ai.completion"), str
|
|
508
|
+
):
|
|
509
|
+
msgs = []
|
|
510
|
+
if attrs.get("gen_ai.prompt"):
|
|
511
|
+
msgs.append({"role": "user", "content": attrs["gen_ai.prompt"]})
|
|
512
|
+
if attrs.get("gen_ai.completion"):
|
|
513
|
+
msgs.append({"role": "assistant", "content": attrs["gen_ai.completion"]})
|
|
514
|
+
return msgs, system_text
|
|
515
|
+
return _event_messages(span), system_text
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
_OTEL_META_SUFFIXES = {
|
|
519
|
+
"route",
|
|
520
|
+
"endpoint",
|
|
521
|
+
"feature",
|
|
522
|
+
"feedback",
|
|
523
|
+
"user_feedback",
|
|
524
|
+
"rating",
|
|
525
|
+
"intent",
|
|
526
|
+
}
|
|
527
|
+
|
|
528
|
+
|
|
529
|
+
def parse_otel_trace(trace_id: str, spans: list[dict[str, Any]], file: str, index: int) -> Trace:
|
|
530
|
+
def end_time(s: dict[str, Any]) -> int:
|
|
531
|
+
try:
|
|
532
|
+
return int(s.get("endTimeUnixNano") or s.get("end_time_unix_nano") or 0)
|
|
533
|
+
except (TypeError, ValueError):
|
|
534
|
+
return 0
|
|
535
|
+
|
|
536
|
+
spans = sorted(spans, key=end_time)
|
|
537
|
+
tools: list[str] = []
|
|
538
|
+
llm_spans = []
|
|
539
|
+
md: dict[str, Any] = {}
|
|
540
|
+
model = None
|
|
541
|
+
route = None
|
|
542
|
+
error = False
|
|
543
|
+
for s in spans:
|
|
544
|
+
attrs = _otel_attrs(s.get("attributes"))
|
|
545
|
+
op = attrs.get("gen_ai.operation.name")
|
|
546
|
+
if op == "execute_tool" and attrs.get("gen_ai.tool.name"):
|
|
547
|
+
tools.append(str(attrs["gen_ai.tool.name"]))
|
|
548
|
+
if any(k.startswith("gen_ai.") for k in attrs) and op not in ("execute_tool", "embeddings"):
|
|
549
|
+
llm_spans.append(s)
|
|
550
|
+
error = error or _span_is_error(s)
|
|
551
|
+
parent = s.get("parentSpanId") or s.get("parent_span_id") or s.get("parent_id")
|
|
552
|
+
if not parent:
|
|
553
|
+
route = route or attrs.get("gen_ai.agent.name") or s.get("name")
|
|
554
|
+
for k, v in attrs.items():
|
|
555
|
+
if not k.startswith("gen_ai.") and isinstance(v, (str, int, float, bool)):
|
|
556
|
+
md[k] = v
|
|
557
|
+
for k, v in attrs.items():
|
|
558
|
+
short = k.split(".")[-1]
|
|
559
|
+
if short in _OTEL_META_SUFFIXES and not k.startswith("gen_ai.") and short not in md:
|
|
560
|
+
md[short] = v
|
|
561
|
+
res = s.get("_resource") or {}
|
|
562
|
+
if res.get("service.name"):
|
|
563
|
+
md.setdefault("service.name", res["service.name"])
|
|
564
|
+
messages: list[dict[str, Any]] = []
|
|
565
|
+
system = None
|
|
566
|
+
for s in reversed(llm_spans):
|
|
567
|
+
attrs = _otel_attrs(s.get("attributes"))
|
|
568
|
+
msgs, system = _span_messages(s, tools)
|
|
569
|
+
if msgs:
|
|
570
|
+
messages = msgs
|
|
571
|
+
model = attrs.get("gen_ai.response.model") or attrs.get("gen_ai.request.model")
|
|
572
|
+
provider = attrs.get("gen_ai.provider.name") or attrs.get("gen_ai.system")
|
|
573
|
+
if provider:
|
|
574
|
+
md["provider"] = provider
|
|
575
|
+
break
|
|
576
|
+
if not messages:
|
|
577
|
+
raise SkipRecord("no gen_ai spans with message content (content capture may be off)")
|
|
578
|
+
if isinstance(md.get("route"), str):
|
|
579
|
+
route = md["route"]
|
|
580
|
+
return build_trace(
|
|
581
|
+
messages,
|
|
582
|
+
fmt="otel",
|
|
583
|
+
file=file,
|
|
584
|
+
index=index,
|
|
585
|
+
given_id=trace_id or None,
|
|
586
|
+
system=system,
|
|
587
|
+
tools=tools,
|
|
588
|
+
metadata=md,
|
|
589
|
+
error=error,
|
|
590
|
+
route=str(route) if route else None,
|
|
591
|
+
model=model if isinstance(model, str) else None,
|
|
592
|
+
)
|
|
593
|
+
|
|
594
|
+
|
|
595
|
+
# --------------------------------------------------------------------------- generic
|
|
596
|
+
|
|
597
|
+
|
|
598
|
+
def parse_generic(rec: dict[str, Any], file: str, index: int) -> Trace:
|
|
599
|
+
inp = next((rec[k] for k in ("input", "prompt", "question") if k in rec), None)
|
|
600
|
+
out = next((rec[k] for k in ("output", "completion", "answer", "response") if k in rec), None)
|
|
601
|
+
if inp is None:
|
|
602
|
+
raise SkipRecord("no input field")
|
|
603
|
+
messages = _lf_messages(inp, "user")
|
|
604
|
+
if out is not None:
|
|
605
|
+
out_msgs = _lf_messages(out, "assistant")
|
|
606
|
+
if out_msgs:
|
|
607
|
+
messages.append({"role": "assistant", "content": out_msgs[-1].get("content")})
|
|
608
|
+
md = dict(rec.get("metadata") or {}) if isinstance(rec.get("metadata"), dict) else {}
|
|
609
|
+
md.update(
|
|
610
|
+
_scalar_extras(
|
|
611
|
+
rec,
|
|
612
|
+
{
|
|
613
|
+
"input",
|
|
614
|
+
"prompt",
|
|
615
|
+
"question",
|
|
616
|
+
"output",
|
|
617
|
+
"completion",
|
|
618
|
+
"answer",
|
|
619
|
+
"response",
|
|
620
|
+
"metadata",
|
|
621
|
+
"id",
|
|
622
|
+
"tools",
|
|
623
|
+
},
|
|
624
|
+
)
|
|
625
|
+
)
|
|
626
|
+
tools = [str(t) for t in rec.get("tools") or md.get("tools") or [] if isinstance(t, str)]
|
|
627
|
+
if isinstance(md.get("tool"), str):
|
|
628
|
+
tools.append(md["tool"])
|
|
629
|
+
return build_trace(
|
|
630
|
+
messages,
|
|
631
|
+
fmt="generic",
|
|
632
|
+
file=file,
|
|
633
|
+
index=index,
|
|
634
|
+
given_id=rec.get("id"),
|
|
635
|
+
tools=tools,
|
|
636
|
+
metadata=md,
|
|
637
|
+
)
|
|
638
|
+
|
|
639
|
+
|
|
640
|
+
PARSERS = {
|
|
641
|
+
"openai": parse_openai,
|
|
642
|
+
"anthropic": parse_anthropic,
|
|
643
|
+
"langfuse": parse_langfuse,
|
|
644
|
+
"generic": parse_generic,
|
|
645
|
+
}
|