eval-builder 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,645 @@
1
+ """Parsers for each supported log format. Each yields normalized Trace objects."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from collections.abc import Iterator
7
+ from typing import Any
8
+
9
+ from ..schema import FEEDBACK_KEYS, Trace, build_trace, normalize_feedback, text_of
10
+
11
+
12
+ class SkipRecord(Exception):
13
+ """Raised when a record cannot become a trace; the message is the reason."""
14
+
15
+
16
+ FORMATS = ("openai", "anthropic", "langfuse", "otel", "generic")
17
+
18
+
19
+ # --------------------------------------------------------------------------- detection
20
+
21
+
22
+ def detect_format(records: list[Any]) -> str:
23
+ """Guess the format from the first few records. Raises ValueError if unsure."""
24
+ votes: dict[str, int] = {}
25
+ for rec in records[:20]:
26
+ fmt = _detect_one(rec)
27
+ if fmt:
28
+ votes[fmt] = votes.get(fmt, 0) + 1
29
+ if not votes:
30
+ raise ValueError(
31
+ "could not detect the log format; pass --format "
32
+ "(openai, anthropic, langfuse, otel, generic)"
33
+ )
34
+ return max(votes.items(), key=lambda kv: kv[1])[0]
35
+
36
+
37
+ def _detect_one(rec: Any) -> str | None:
38
+ if not isinstance(rec, dict):
39
+ return None
40
+ if (
41
+ "resourceSpans" in rec
42
+ or "spanId" in rec
43
+ or "span_id" in rec
44
+ or (isinstance(rec.get("context"), dict) and "span_id" in rec["context"])
45
+ ):
46
+ return "otel"
47
+ if "observations" in rec or "htmlPath" in rec or "projectId" in rec:
48
+ return "langfuse"
49
+ if _looks_anthropic(rec):
50
+ return "anthropic"
51
+ if (
52
+ "messages" in rec
53
+ or "choices" in rec
54
+ or (isinstance(rec.get("request"), dict) and "messages" in rec["request"])
55
+ ):
56
+ return "openai"
57
+ if "input" in rec and "output" in rec and isinstance(rec.get("scores"), list):
58
+ return "langfuse"
59
+ if any(k in rec for k in ("input", "prompt", "question")) and any(
60
+ k in rec for k in ("output", "completion", "answer", "response")
61
+ ):
62
+ return "generic"
63
+ return None
64
+
65
+
66
+ def _looks_anthropic(rec: dict[str, Any]) -> bool:
67
+ req = _sub(rec, "request")
68
+ resp = rec.get("response")
69
+ if (
70
+ isinstance(resp, dict)
71
+ and isinstance(resp.get("content"), list)
72
+ and ("stop_reason" in resp or resp.get("type") == "message")
73
+ ):
74
+ return True
75
+ if rec.get("type") == "message" and isinstance(rec.get("content"), list):
76
+ return True
77
+ if isinstance(req, dict) and "messages" in req:
78
+ if isinstance(req.get("system"), (str, list)) and "choices" not in rec:
79
+ return True
80
+ for m in req.get("messages") or []:
81
+ content = m.get("content") if isinstance(m, dict) else None
82
+ if isinstance(content, list) and any(
83
+ isinstance(b, dict) and b.get("type") in ("tool_use", "tool_result")
84
+ for b in content
85
+ ):
86
+ return True
87
+ return False
88
+
89
+
90
+ # --------------------------------------------------------------------------- helpers
91
+
92
+ _RESERVED_OPENAI = {
93
+ "messages",
94
+ "request",
95
+ "response",
96
+ "choices",
97
+ "metadata",
98
+ "id",
99
+ "model",
100
+ "usage",
101
+ "object",
102
+ "created",
103
+ "system_fingerprint",
104
+ "tools",
105
+ }
106
+
107
+
108
+ def _scalar_extras(rec: dict[str, Any], reserved: set[str]) -> dict[str, Any]:
109
+ out: dict[str, Any] = {}
110
+ for k, v in rec.items():
111
+ if k in reserved:
112
+ continue
113
+ if (
114
+ isinstance(v, (str, int, float, bool))
115
+ or v is None
116
+ or isinstance(v, list)
117
+ and all(isinstance(x, (str, int, float)) for x in v)
118
+ ):
119
+ out[k] = v
120
+ return out
121
+
122
+
123
+ def _sub(rec: dict[str, Any], key: str) -> dict[str, Any]:
124
+ """rec[key] when it is a dict, else rec itself (flat records)."""
125
+ v = rec.get(key)
126
+ return v if isinstance(v, dict) else rec
127
+
128
+
129
+ def _dict_or_empty(v: Any) -> dict[str, Any]:
130
+ return v if isinstance(v, dict) else {}
131
+
132
+
133
+ def _openai_tool_names(msg: dict[str, Any]) -> list[str]:
134
+ names = []
135
+ for tc in msg.get("tool_calls") or []:
136
+ if isinstance(tc, dict):
137
+ fn = tc.get("function") or {}
138
+ name = fn.get("name") or tc.get("name")
139
+ if name:
140
+ names.append(str(name))
141
+ fc = msg.get("function_call")
142
+ if isinstance(fc, dict) and fc.get("name"):
143
+ names.append(str(fc["name"]))
144
+ return names
145
+
146
+
147
+ # --------------------------------------------------------------------------- openai
148
+
149
+
150
+ def parse_openai(rec: dict[str, Any], file: str, index: int) -> Trace:
151
+ req = _sub(rec, "request")
152
+ messages = list(req.get("messages") or [])
153
+ resp = _sub(rec, "response")
154
+ choices = resp.get("choices") if isinstance(resp, dict) else None
155
+ if isinstance(choices, list) and choices:
156
+ msg = choices[0].get("message") or {"role": "assistant", "content": choices[0].get("text")}
157
+ messages.append(msg)
158
+ if not messages:
159
+ raise SkipRecord("no messages")
160
+ tools: list[str] = []
161
+ for m in messages:
162
+ if isinstance(m, dict):
163
+ tools.extend(_openai_tool_names(m))
164
+ md = dict(rec.get("metadata") or {}) if isinstance(rec.get("metadata"), dict) else {}
165
+ md.update(_scalar_extras(rec, _RESERVED_OPENAI))
166
+ model = rec.get("model") or req.get("model") or (resp.get("model") if resp else None)
167
+ err = bool(rec.get("error")) or bool(isinstance(resp, dict) and resp.get("error"))
168
+ return build_trace(
169
+ [m for m in messages if isinstance(m, dict)],
170
+ fmt="openai",
171
+ file=file,
172
+ index=index,
173
+ given_id=rec.get("id"),
174
+ tools=tools,
175
+ metadata=md,
176
+ model=model if isinstance(model, str) else None,
177
+ error=err,
178
+ )
179
+
180
+
181
+ # --------------------------------------------------------------------------- anthropic
182
+
183
+
184
+ def _anthropic_message(m: dict[str, Any], tools: list[str]) -> dict[str, Any]:
185
+ content = m.get("content")
186
+ role = m.get("role", "user")
187
+ if isinstance(content, list):
188
+ texts = []
189
+ only_tool_results = bool(content)
190
+ for b in content:
191
+ if not isinstance(b, dict):
192
+ texts.append(str(b))
193
+ only_tool_results = False
194
+ continue
195
+ btype = b.get("type")
196
+ if btype == "tool_use":
197
+ if b.get("name"):
198
+ tools.append(str(b["name"]))
199
+ continue
200
+ if btype == "tool_result":
201
+ texts.append(text_of(b.get("content")))
202
+ continue
203
+ only_tool_results = False
204
+ if btype in ("text", None):
205
+ texts.append(text_of(b))
206
+ if role == "user" and only_tool_results:
207
+ role = "tool"
208
+ return {"role": role, "content": "\n".join(t for t in texts if t)}
209
+ return {"role": role, "content": text_of(content)}
210
+
211
+
212
+ def parse_anthropic(rec: dict[str, Any], file: str, index: int) -> Trace:
213
+ req = _sub(rec, "request")
214
+ tools: list[str] = []
215
+ messages = [
216
+ _anthropic_message(m, tools) for m in req.get("messages") or [] if isinstance(m, dict)
217
+ ]
218
+ resp = rec.get("response") if isinstance(rec.get("response"), dict) else None
219
+ if resp is None and rec.get("type") == "message" and "content" in rec and rec is not req:
220
+ resp = rec
221
+ if resp is not None and isinstance(resp.get("content"), list):
222
+ messages.append(
223
+ _anthropic_message({"role": "assistant", "content": resp["content"]}, tools)
224
+ )
225
+ if not messages:
226
+ raise SkipRecord("no messages")
227
+ system = req.get("system")
228
+ md = dict(rec.get("metadata") or {}) if isinstance(rec.get("metadata"), dict) else {}
229
+ md.update(_scalar_extras(rec, _RESERVED_OPENAI | {"system", "type", "content", "stop_reason"}))
230
+ if resp is not None and resp.get("stop_reason"):
231
+ md["stop_reason"] = resp["stop_reason"]
232
+ model = req.get("model") or (resp or {}).get("model")
233
+ err = bool(rec.get("error")) or (resp is not None and resp.get("type") == "error")
234
+ return build_trace(
235
+ messages,
236
+ fmt="anthropic",
237
+ file=file,
238
+ index=index,
239
+ given_id=rec.get("id") or (resp or {}).get("id"),
240
+ system=text_of(system) if system else None,
241
+ tools=tools,
242
+ metadata=md,
243
+ model=model if isinstance(model, str) else None,
244
+ error=err,
245
+ )
246
+
247
+
248
+ # --------------------------------------------------------------------------- langfuse
249
+
250
+ _TEXT_KEYS = (
251
+ "input",
252
+ "query",
253
+ "question",
254
+ "prompt",
255
+ "text",
256
+ "content",
257
+ "message",
258
+ "output",
259
+ "answer",
260
+ "completion",
261
+ "response",
262
+ "result",
263
+ )
264
+
265
+
266
+ def _lf_messages(value: Any, default_role: str) -> list[dict[str, Any]]:
267
+ if value is None:
268
+ return []
269
+ if (
270
+ isinstance(value, list)
271
+ and value
272
+ and all(isinstance(m, dict) and "role" in m for m in value)
273
+ ):
274
+ return value
275
+ if isinstance(value, dict):
276
+ if isinstance(value.get("messages"), list):
277
+ return _lf_messages(value["messages"], default_role)
278
+ if "role" in value and "content" in value:
279
+ return [value]
280
+ if isinstance(value.get("choices"), list) and value["choices"]:
281
+ msg = value["choices"][0].get("message")
282
+ if isinstance(msg, dict):
283
+ return [msg]
284
+ for key in _TEXT_KEYS:
285
+ if isinstance(value.get(key), str):
286
+ return [{"role": default_role, "content": value[key]}]
287
+ return [{"role": default_role, "content": json.dumps(value, ensure_ascii=False)}]
288
+ if isinstance(value, str):
289
+ return [{"role": default_role, "content": value}]
290
+ return [{"role": default_role, "content": text_of(value)}]
291
+
292
+
293
+ def parse_langfuse(rec: dict[str, Any], file: str, index: int) -> Trace:
294
+ observations = [o for o in rec.get("observations") or [] if isinstance(o, dict)]
295
+ generations = [o for o in observations if o.get("type") == "GENERATION"]
296
+ generations.sort(key=lambda o: str(o.get("startTime") or ""))
297
+ inp, out = rec.get("input"), rec.get("output")
298
+ model = None
299
+ tools: list[str] = []
300
+ if generations:
301
+ last = generations[-1]
302
+ model = last.get("model")
303
+ if inp is None:
304
+ inp = last.get("input")
305
+ if out is None:
306
+ out = last.get("output")
307
+ for o in observations:
308
+ if o.get("type") == "TOOL" and o.get("name"):
309
+ tools.append(str(o["name"]))
310
+ if o.get("type") == "GENERATION":
311
+ for m in _lf_messages(o.get("output"), "assistant"):
312
+ tools.extend(_openai_tool_names(m))
313
+ messages = _lf_messages(inp, "user")
314
+ out_msgs = _lf_messages(out, "assistant")
315
+ if out_msgs:
316
+ final = out_msgs[-1]
317
+ messages = messages + [{"role": "assistant", "content": final.get("content")}]
318
+ tools.extend(_openai_tool_names(final))
319
+ if not messages:
320
+ raise SkipRecord("no input or output")
321
+ error = any(str(o.get("level", "")).upper() == "ERROR" for o in observations)
322
+ if str(rec.get("level", "")).upper() == "ERROR":
323
+ error = True
324
+ feedback = None
325
+ other_scores: dict[str, Any] = {}
326
+ for s in rec.get("scores") or []:
327
+ if not isinstance(s, dict):
328
+ continue
329
+ name = str(s.get("name", "")).lower().replace("-", "_").replace(" ", "_")
330
+ value = s.get("value", s.get("stringValue"))
331
+ if name in FEEDBACK_KEYS or "feedback" in name or "thumb" in name:
332
+ feedback = feedback or normalize_feedback(value)
333
+ else:
334
+ other_scores[s.get("name", "score")] = value
335
+ md = dict(rec.get("metadata") or {}) if isinstance(rec.get("metadata"), dict) else {}
336
+ if rec.get("tags"):
337
+ md["tags"] = rec["tags"]
338
+ for key in ("release", "version", "environment"):
339
+ if rec.get(key):
340
+ md[key] = rec[key]
341
+ if other_scores:
342
+ md["scores"] = other_scores
343
+ return build_trace(
344
+ messages,
345
+ fmt="langfuse",
346
+ file=file,
347
+ index=index,
348
+ given_id=rec.get("id"),
349
+ tools=tools,
350
+ metadata=md,
351
+ error=error,
352
+ feedback=feedback,
353
+ route=str(rec["name"]) if rec.get("name") else None,
354
+ model=model if isinstance(model, str) else None,
355
+ )
356
+
357
+
358
+ # --------------------------------------------------------------------------- otel
359
+
360
+
361
+ def _otel_value(v: Any) -> Any:
362
+ if not isinstance(v, dict):
363
+ return v
364
+ for key in ("stringValue", "boolValue", "doubleValue"):
365
+ if key in v:
366
+ return v[key]
367
+ if "intValue" in v:
368
+ try:
369
+ return int(v["intValue"])
370
+ except (TypeError, ValueError):
371
+ return v["intValue"]
372
+ if "arrayValue" in v:
373
+ return [_otel_value(x) for x in (v["arrayValue"] or {}).get("values", [])]
374
+ if "kvlistValue" in v:
375
+ return {
376
+ kv["key"]: _otel_value(kv.get("value")) for kv in v["kvlistValue"].get("values", [])
377
+ }
378
+ return v
379
+
380
+
381
+ def _otel_attrs(attrs: Any) -> dict[str, Any]:
382
+ if isinstance(attrs, dict):
383
+ return dict(attrs)
384
+ out: dict[str, Any] = {}
385
+ for kv in attrs or []:
386
+ if isinstance(kv, dict) and "key" in kv:
387
+ out[kv["key"]] = _otel_value(kv.get("value"))
388
+ return out
389
+
390
+
391
+ def iter_otel_spans(rec: dict[str, Any]) -> Iterator[dict[str, Any]]:
392
+ """Flatten OTLP JSON (resourceSpans) or pass through a single flat span."""
393
+ if "resourceSpans" in rec:
394
+ for rs in rec.get("resourceSpans") or []:
395
+ res_attrs = _otel_attrs((rs.get("resource") or {}).get("attributes"))
396
+ for ss in rs.get("scopeSpans") or rs.get("instrumentationLibrarySpans") or []:
397
+ for span in ss.get("spans") or []:
398
+ yield {**span, "_resource": res_attrs}
399
+ else:
400
+ yield rec
401
+
402
+
403
+ def _span_trace_id(span: dict[str, Any]) -> str:
404
+ ctx = _dict_or_empty(span.get("context"))
405
+ return str(span.get("traceId") or span.get("trace_id") or ctx.get("trace_id") or "")
406
+
407
+
408
+ def _span_is_error(span: dict[str, Any]) -> bool:
409
+ status = span.get("status") or {}
410
+ code = status.get("code", status.get("status_code"))
411
+ return code in (2, "2", "STATUS_CODE_ERROR", "ERROR") or bool(
412
+ _otel_attrs(span.get("attributes")).get("error.type")
413
+ )
414
+
415
+
416
+ def _parse_json_attr(v: Any) -> Any:
417
+ if isinstance(v, str):
418
+ try:
419
+ return json.loads(v)
420
+ except ValueError:
421
+ return v
422
+ return v
423
+
424
+
425
+ def _semconv_messages(raw: Any, tools: list[str]) -> list[dict[str, Any]]:
426
+ """gen_ai.input.messages / gen_ai.output.messages: [{role, parts:[{type, content}]}]."""
427
+ raw = _parse_json_attr(raw)
428
+ msgs: list[dict[str, Any]] = []
429
+ if not isinstance(raw, list):
430
+ return msgs
431
+ for m in raw:
432
+ if not isinstance(m, dict):
433
+ continue
434
+ parts = m.get("parts")
435
+ if isinstance(parts, list):
436
+ texts = []
437
+ for p in parts:
438
+ if not isinstance(p, dict):
439
+ continue
440
+ if p.get("type") == "tool_call":
441
+ if p.get("name"):
442
+ tools.append(str(p["name"]))
443
+ continue
444
+ if p.get("type") in ("text", "tool_call_response", None):
445
+ texts.append(text_of(p.get("content", p.get("response", p.get("result")))))
446
+ content = "\n".join(t for t in texts if t)
447
+ else:
448
+ content = text_of(m.get("content"))
449
+ role = m.get("role", "user")
450
+ msgs.append({"role": "tool" if role == "tool" else role, "content": content})
451
+ return msgs
452
+
453
+
454
+ def _indexed_messages(attrs: dict[str, Any], prefix: str) -> list[dict[str, Any]]:
455
+ """OpenLLMetry style gen_ai.prompt.0.role / gen_ai.prompt.0.content."""
456
+ idx: dict[int, dict[str, Any]] = {}
457
+ for k, v in attrs.items():
458
+ if not k.startswith(prefix + "."):
459
+ continue
460
+ rest = k[len(prefix) + 1 :].split(".", 1)
461
+ if len(rest) == 2 and rest[0].isdigit() and rest[1] in ("role", "content"):
462
+ idx.setdefault(int(rest[0]), {})[rest[1]] = v
463
+ return [idx[i] for i in sorted(idx)]
464
+
465
+
466
+ def _event_messages(span: dict[str, Any]) -> list[dict[str, Any]]:
467
+ msgs = []
468
+ for ev in span.get("events") or []:
469
+ name = ev.get("name", "")
470
+ attrs = _otel_attrs(ev.get("attributes"))
471
+ body = _parse_json_attr(ev.get("body", attrs.get("body")))
472
+ payload = body if isinstance(body, dict) else attrs
473
+ if name in (
474
+ "gen_ai.system.message",
475
+ "gen_ai.user.message",
476
+ "gen_ai.assistant.message",
477
+ "gen_ai.tool.message",
478
+ ):
479
+ role = name.split(".")[1]
480
+ msgs.append({"role": role, "content": text_of(payload.get("content"))})
481
+ elif name == "gen_ai.choice":
482
+ message = _dict_or_empty(payload.get("message")) or payload
483
+ msgs.append({"role": "assistant", "content": text_of(message.get("content"))})
484
+ return msgs
485
+
486
+
487
+ def _span_messages(
488
+ span: dict[str, Any], tools: list[str]
489
+ ) -> tuple[list[dict[str, Any]], str | None]:
490
+ attrs = _otel_attrs(span.get("attributes"))
491
+ system = attrs.get("gen_ai.system_instructions")
492
+ system_text = None
493
+ if system is not None:
494
+ parsed = _parse_json_attr(system)
495
+ system_text = text_of(parsed) or (system if isinstance(system, str) else None)
496
+ if "gen_ai.input.messages" in attrs or "gen_ai.output.messages" in attrs:
497
+ msgs = _semconv_messages(attrs.get("gen_ai.input.messages"), tools)
498
+ msgs += _semconv_messages(attrs.get("gen_ai.output.messages"), tools)
499
+ return msgs, system_text
500
+ prompt = _indexed_messages(attrs, "gen_ai.prompt")
501
+ completion = _indexed_messages(attrs, "gen_ai.completion")
502
+ if prompt or completion:
503
+ return prompt + [
504
+ {"role": m.get("role", "assistant"), "content": m.get("content")} for m in completion
505
+ ], system_text
506
+ if isinstance(attrs.get("gen_ai.prompt"), str) or isinstance(
507
+ attrs.get("gen_ai.completion"), str
508
+ ):
509
+ msgs = []
510
+ if attrs.get("gen_ai.prompt"):
511
+ msgs.append({"role": "user", "content": attrs["gen_ai.prompt"]})
512
+ if attrs.get("gen_ai.completion"):
513
+ msgs.append({"role": "assistant", "content": attrs["gen_ai.completion"]})
514
+ return msgs, system_text
515
+ return _event_messages(span), system_text
516
+
517
+
518
+ _OTEL_META_SUFFIXES = {
519
+ "route",
520
+ "endpoint",
521
+ "feature",
522
+ "feedback",
523
+ "user_feedback",
524
+ "rating",
525
+ "intent",
526
+ }
527
+
528
+
529
+ def parse_otel_trace(trace_id: str, spans: list[dict[str, Any]], file: str, index: int) -> Trace:
530
+ def end_time(s: dict[str, Any]) -> int:
531
+ try:
532
+ return int(s.get("endTimeUnixNano") or s.get("end_time_unix_nano") or 0)
533
+ except (TypeError, ValueError):
534
+ return 0
535
+
536
+ spans = sorted(spans, key=end_time)
537
+ tools: list[str] = []
538
+ llm_spans = []
539
+ md: dict[str, Any] = {}
540
+ model = None
541
+ route = None
542
+ error = False
543
+ for s in spans:
544
+ attrs = _otel_attrs(s.get("attributes"))
545
+ op = attrs.get("gen_ai.operation.name")
546
+ if op == "execute_tool" and attrs.get("gen_ai.tool.name"):
547
+ tools.append(str(attrs["gen_ai.tool.name"]))
548
+ if any(k.startswith("gen_ai.") for k in attrs) and op not in ("execute_tool", "embeddings"):
549
+ llm_spans.append(s)
550
+ error = error or _span_is_error(s)
551
+ parent = s.get("parentSpanId") or s.get("parent_span_id") or s.get("parent_id")
552
+ if not parent:
553
+ route = route or attrs.get("gen_ai.agent.name") or s.get("name")
554
+ for k, v in attrs.items():
555
+ if not k.startswith("gen_ai.") and isinstance(v, (str, int, float, bool)):
556
+ md[k] = v
557
+ for k, v in attrs.items():
558
+ short = k.split(".")[-1]
559
+ if short in _OTEL_META_SUFFIXES and not k.startswith("gen_ai.") and short not in md:
560
+ md[short] = v
561
+ res = s.get("_resource") or {}
562
+ if res.get("service.name"):
563
+ md.setdefault("service.name", res["service.name"])
564
+ messages: list[dict[str, Any]] = []
565
+ system = None
566
+ for s in reversed(llm_spans):
567
+ attrs = _otel_attrs(s.get("attributes"))
568
+ msgs, system = _span_messages(s, tools)
569
+ if msgs:
570
+ messages = msgs
571
+ model = attrs.get("gen_ai.response.model") or attrs.get("gen_ai.request.model")
572
+ provider = attrs.get("gen_ai.provider.name") or attrs.get("gen_ai.system")
573
+ if provider:
574
+ md["provider"] = provider
575
+ break
576
+ if not messages:
577
+ raise SkipRecord("no gen_ai spans with message content (content capture may be off)")
578
+ if isinstance(md.get("route"), str):
579
+ route = md["route"]
580
+ return build_trace(
581
+ messages,
582
+ fmt="otel",
583
+ file=file,
584
+ index=index,
585
+ given_id=trace_id or None,
586
+ system=system,
587
+ tools=tools,
588
+ metadata=md,
589
+ error=error,
590
+ route=str(route) if route else None,
591
+ model=model if isinstance(model, str) else None,
592
+ )
593
+
594
+
595
+ # --------------------------------------------------------------------------- generic
596
+
597
+
598
+ def parse_generic(rec: dict[str, Any], file: str, index: int) -> Trace:
599
+ inp = next((rec[k] for k in ("input", "prompt", "question") if k in rec), None)
600
+ out = next((rec[k] for k in ("output", "completion", "answer", "response") if k in rec), None)
601
+ if inp is None:
602
+ raise SkipRecord("no input field")
603
+ messages = _lf_messages(inp, "user")
604
+ if out is not None:
605
+ out_msgs = _lf_messages(out, "assistant")
606
+ if out_msgs:
607
+ messages.append({"role": "assistant", "content": out_msgs[-1].get("content")})
608
+ md = dict(rec.get("metadata") or {}) if isinstance(rec.get("metadata"), dict) else {}
609
+ md.update(
610
+ _scalar_extras(
611
+ rec,
612
+ {
613
+ "input",
614
+ "prompt",
615
+ "question",
616
+ "output",
617
+ "completion",
618
+ "answer",
619
+ "response",
620
+ "metadata",
621
+ "id",
622
+ "tools",
623
+ },
624
+ )
625
+ )
626
+ tools = [str(t) for t in rec.get("tools") or md.get("tools") or [] if isinstance(t, str)]
627
+ if isinstance(md.get("tool"), str):
628
+ tools.append(md["tool"])
629
+ return build_trace(
630
+ messages,
631
+ fmt="generic",
632
+ file=file,
633
+ index=index,
634
+ given_id=rec.get("id"),
635
+ tools=tools,
636
+ metadata=md,
637
+ )
638
+
639
+
640
+ PARSERS = {
641
+ "openai": parse_openai,
642
+ "anthropic": parse_anthropic,
643
+ "langfuse": parse_langfuse,
644
+ "generic": parse_generic,
645
+ }