loopview 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- loopview/__init__.py +3 -0
- loopview/app.py +252 -0
- loopview/cli.py +128 -0
- loopview/cost/__init__.py +1 -0
- loopview/cost/pricing.json +62 -0
- loopview/cost/pricing.py +74 -0
- loopview/cost/split.py +228 -0
- loopview/demo_data/flagship.otlp.jsonl +13 -0
- loopview/devtools/__init__.py +1 -0
- loopview/devtools/dump_normalized.py +211 -0
- loopview/ingest/__init__.py +1 -0
- loopview/ingest/otlp.py +305 -0
- loopview/ingest/raw.py +47 -0
- loopview/live.py +154 -0
- loopview/normalize/__init__.py +1 -0
- loopview/normalize/adapters/__init__.py +14 -0
- loopview/normalize/adapters/base.py +87 -0
- loopview/normalize/adapters/gen_ai.py +252 -0
- loopview/normalize/adapters/generic.py +22 -0
- loopview/normalize/adapters/openinference.py +367 -0
- loopview/normalize/derived_tools.py +90 -0
- loopview/normalize/loop_nodes.py +193 -0
- loopview/normalize/normalizer.py +326 -0
- loopview/normalize/schema.py +161 -0
- loopview/normalize/transitions.py +106 -0
- loopview/py.typed +0 -0
- loopview/static/assets/elk-worker.min-DfmSo98M.js +22 -0
- loopview/static/assets/index-CEGppBgk.css +1 -0
- loopview/static/assets/index-DzC96LiQ.js +21 -0
- loopview/static/assets/inter-cyrillic-ext-wght-normal-BOeWTOD4.woff2 +0 -0
- loopview/static/assets/inter-cyrillic-wght-normal-DqGufNeO.woff2 +0 -0
- loopview/static/assets/inter-greek-ext-wght-normal-DlzME5K_.woff2 +0 -0
- loopview/static/assets/inter-greek-wght-normal-CkhJZR-_.woff2 +0 -0
- loopview/static/assets/inter-latin-ext-wght-normal-DO1Apj_S.woff2 +0 -0
- loopview/static/assets/inter-latin-wght-normal-Dx4kXJAl.woff2 +0 -0
- loopview/static/assets/inter-vietnamese-wght-normal-CBcvBZtf.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-cyrillic-wght-normal-D73BlboJ.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-greek-wght-normal-Bw9x6K1M.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-latin-ext-wght-normal-DBQx-q_a.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-latin-wght-normal-B9CIFXIH.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-vietnamese-wght-normal-Bt-aOZkq.woff2 +0 -0
- loopview/static/index.html +13 -0
- loopview/store/__init__.py +1 -0
- loopview/store/capture.py +112 -0
- loopview/store/memory.py +165 -0
- loopview/tools/__init__.py +1 -0
- loopview/tools/report.py +347 -0
- loopview-0.1.0.dist-info/METADATA +72 -0
- loopview-0.1.0.dist-info/RECORD +51 -0
- loopview-0.1.0.dist-info/WHEEL +4 -0
- loopview-0.1.0.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,367 @@
|
|
|
1
|
+
"""Adapter for OpenInference (Arize) semantic conventions.
|
|
2
|
+
|
|
3
|
+
Written against the spec in Arize-ai/openinference (spec/semantic_conventions.md)
|
|
4
|
+
as emitted by openinference-instrumentation-langchain 0.1.76 and
|
|
5
|
+
openinference-semantic-conventions 0.1.39. The spec itself has no version number.
|
|
6
|
+
|
|
7
|
+
A span is OpenInference if it has openinference.span.kind.
|
|
8
|
+
|
|
9
|
+
LangGraph: the LangChain instrumentor copies LangGraph's run metadata into the
|
|
10
|
+
`metadata` attribute. That is how we tell a real graph node (langgraph_node ==
|
|
11
|
+
span name) from plumbing inside a node, such as a routing function or an output
|
|
12
|
+
parser (langgraph_node != span name), which we keep but hide.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import re
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
from loopview.ingest.raw import RawSpan
|
|
19
|
+
from loopview.normalize.adapters.base import (
|
|
20
|
+
Classification,
|
|
21
|
+
ParentHint,
|
|
22
|
+
as_int,
|
|
23
|
+
maybe_json,
|
|
24
|
+
)
|
|
25
|
+
from loopview.normalize.schema import Message, MessagePart, ModelCall, ToolCall, Usage
|
|
26
|
+
|
|
27
|
+
TOOL_LIKE_KINDS = {
|
|
28
|
+
"TOOL": "tool",
|
|
29
|
+
"RETRIEVER": "retriever",
|
|
30
|
+
"EMBEDDING": "embedding",
|
|
31
|
+
"RERANKER": "reranker",
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class OpenInferenceAdapter:
|
|
36
|
+
name = "openinference"
|
|
37
|
+
|
|
38
|
+
def matches(self, span: RawSpan) -> bool:
|
|
39
|
+
return "openinference.span.kind" in span.attributes
|
|
40
|
+
|
|
41
|
+
def classify(self, span: RawSpan) -> Classification:
|
|
42
|
+
attrs = span.attributes
|
|
43
|
+
kind = str(attrs["openinference.span.kind"]).upper()
|
|
44
|
+
io = {
|
|
45
|
+
"input": maybe_json(attrs.get("input.value")),
|
|
46
|
+
"output": maybe_json(attrs.get("output.value")),
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
if kind == "LLM":
|
|
50
|
+
model = attrs.get("llm.model_name") or span.name
|
|
51
|
+
return Classification("model_call", str(model), "llm", model=_model_call(attrs))
|
|
52
|
+
if kind in TOOL_LIKE_KINDS:
|
|
53
|
+
name = attrs.get("tool.name") or span.name
|
|
54
|
+
return Classification(
|
|
55
|
+
"tool_call",
|
|
56
|
+
str(name),
|
|
57
|
+
TOOL_LIKE_KINDS[kind],
|
|
58
|
+
tool=ToolCall(
|
|
59
|
+
name=str(name),
|
|
60
|
+
call_id=attrs.get("tool_call.id"),
|
|
61
|
+
arguments=io["input"],
|
|
62
|
+
result=_unwrap_tool_output(io["output"]),
|
|
63
|
+
),
|
|
64
|
+
)
|
|
65
|
+
if kind == "PROMPT":
|
|
66
|
+
return Classification("node", span.name, "prompt", hidden=True, **io)
|
|
67
|
+
if kind in ("GUARDRAIL", "EVALUATOR"):
|
|
68
|
+
return Classification("node", span.name, kind.lower(), **io)
|
|
69
|
+
|
|
70
|
+
# CHAIN and AGENT: graphs, graph nodes, chains, agents.
|
|
71
|
+
metadata = _metadata(attrs)
|
|
72
|
+
graph_node = metadata.get("langgraph_node")
|
|
73
|
+
if graph_node is None and metadata.get("ls_integration") == "langgraph":
|
|
74
|
+
return Classification("agent", span.name, "graph", **io) # a compiled graph
|
|
75
|
+
if graph_node is not None and graph_node != span.name:
|
|
76
|
+
return Classification("node", span.name, "internal", hidden=True, **io)
|
|
77
|
+
if graph_node == span.name:
|
|
78
|
+
# A real node. If a subgraph runs inside it, the subgraph's own span has
|
|
79
|
+
# the same name and sits directly under it: collapse the two.
|
|
80
|
+
return Classification("node", span.name, "node", collapse_into_parent=True, **io)
|
|
81
|
+
if kind == "AGENT":
|
|
82
|
+
# graph.node.id is the agent's own name where the span name is a method
|
|
83
|
+
# (CrewAI: "Weather assistant._execute_core").
|
|
84
|
+
name = attrs.get("agent.name") or attrs.get("graph.node.id") or _readable(span.name)
|
|
85
|
+
return Classification("agent", str(name), "agent", **io)
|
|
86
|
+
# A chain directly inside a span of the same name is the same step recorded
|
|
87
|
+
# twice (the OpenAI Agents SDK wraps its "Agent workflow" AGENT in a CHAIN).
|
|
88
|
+
return Classification(
|
|
89
|
+
"node", _readable(span.name), "chain", collapse_into_parent=True, **io
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
def infer_parent(self, span: RawSpan) -> ParentHint | None:
|
|
93
|
+
metadata = _metadata(span.attributes)
|
|
94
|
+
graph_node = metadata.get("langgraph_node")
|
|
95
|
+
if graph_node is None:
|
|
96
|
+
return None
|
|
97
|
+
if graph_node != span.name:
|
|
98
|
+
return ParentHint(str(graph_node), "node", "node") # plumbing inside a node
|
|
99
|
+
# A node inside a subgraph: the namespace is "outer:<id>|inner:<id>".
|
|
100
|
+
namespace = str(metadata.get("langgraph_checkpoint_ns", "")).split("|")
|
|
101
|
+
if len(namespace) > 1:
|
|
102
|
+
return ParentHint(namespace[-2].split(":")[0], "agent", "graph")
|
|
103
|
+
return None
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
_UUID_IN_NAME = re.compile(
|
|
107
|
+
r"[_-]?[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}"
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _readable(name: str) -> str:
|
|
112
|
+
"""A span name without the run's UUID in it (CrewAI: "Crew_<uuid>.kickoff")."""
|
|
113
|
+
return _UUID_IN_NAME.sub("", name) or name
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _metadata(attrs: dict[str, Any]) -> dict[str, Any]:
|
|
117
|
+
value = maybe_json(attrs.get("metadata"))
|
|
118
|
+
return value if isinstance(value, dict) else {}
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _unwrap_tool_output(value: Any) -> Any:
|
|
122
|
+
"""LangChain serialises a tool result as a ToolMessage; show just its content."""
|
|
123
|
+
if (
|
|
124
|
+
isinstance(value, dict)
|
|
125
|
+
and value.get("type") == "tool"
|
|
126
|
+
and isinstance(value.get("data"), dict)
|
|
127
|
+
):
|
|
128
|
+
return maybe_json(value["data"].get("content"))
|
|
129
|
+
return value
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _model_call(attrs: dict[str, Any]) -> ModelCall:
|
|
133
|
+
output = _messages(attrs, "llm.output_messages.")
|
|
134
|
+
reasoning = _reasoning_from_raw_output(attrs.get("output.value"))
|
|
135
|
+
if reasoning:
|
|
136
|
+
if not output:
|
|
137
|
+
output = [Message(role="assistant", parts=[])]
|
|
138
|
+
output[0].parts[:0] = [MessagePart(type="reasoning", text=r) for r in reasoning]
|
|
139
|
+
inputs = _messages(attrs, "llm.input_messages.")
|
|
140
|
+
_add_raw_tool_results(inputs, attrs.get("input.value"))
|
|
141
|
+
return ModelCall(
|
|
142
|
+
provider=attrs.get("llm.provider") or attrs.get("llm.system"),
|
|
143
|
+
model=attrs.get("llm.model_name"),
|
|
144
|
+
input=inputs,
|
|
145
|
+
output=output,
|
|
146
|
+
tool_definitions=_tool_definitions(attrs),
|
|
147
|
+
usage=_usage(attrs),
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _tool_definitions(attrs: dict[str, Any]) -> list[Any]:
|
|
152
|
+
"""llm.tools.N.tool.json_schema, in order."""
|
|
153
|
+
return [
|
|
154
|
+
maybe_json(t.get("tool.json_schema"))
|
|
155
|
+
for t in _group_by_index(attrs, "llm.tools.")
|
|
156
|
+
if t.get("tool.json_schema") is not None
|
|
157
|
+
]
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _usage(attrs: dict[str, Any]) -> Usage | None:
|
|
161
|
+
"""Token counts from the spec's attributes, completed from the raw output.
|
|
162
|
+
|
|
163
|
+
The instrumentor leaves out cache attributes that are zero, and doesn't copy
|
|
164
|
+
reasoning (thinking) tokens into an attribute at all. Both are in the raw
|
|
165
|
+
output it also records, as LangChain's usage_metadata."""
|
|
166
|
+
raw = _raw_usage(attrs.get("output.value"))
|
|
167
|
+
usage = Usage(
|
|
168
|
+
input_tokens=as_int(attrs.get("llm.token_count.prompt")),
|
|
169
|
+
output_tokens=as_int(attrs.get("llm.token_count.completion")),
|
|
170
|
+
cache_read_tokens=_first(
|
|
171
|
+
attrs.get("llm.token_count.prompt_details.cache_read"), raw.get("cache_read")
|
|
172
|
+
),
|
|
173
|
+
cache_write_tokens=_first(
|
|
174
|
+
attrs.get("llm.token_count.prompt_details.cache_write"), raw.get("cache_write")
|
|
175
|
+
),
|
|
176
|
+
reasoning_tokens=_first(
|
|
177
|
+
attrs.get("llm.token_count.completion_details.reasoning"), raw.get("reasoning")
|
|
178
|
+
),
|
|
179
|
+
)
|
|
180
|
+
return usage if usage.model_dump(exclude_none=True) else None
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _first(*values: Any) -> int | None:
|
|
184
|
+
for value in values:
|
|
185
|
+
if as_int(value) is not None:
|
|
186
|
+
return as_int(value)
|
|
187
|
+
return None
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def _raw_usage(raw: Any) -> dict[str, int]:
|
|
191
|
+
"""Cache and reasoning counts from LangChain's usage_metadata in the raw output:
|
|
192
|
+
{"input_token_details": {"cache_read", "cache_creation", "ephemeral_5m_input_tokens",
|
|
193
|
+
"ephemeral_1h_input_tokens"}, "output_token_details": {"reasoning"}}.
|
|
194
|
+
|
|
195
|
+
LangChain's cache_creation can read 0 while the per-lifetime counts
|
|
196
|
+
(ephemeral_5m / ephemeral_1h) hold the cache writes, so the larger wins."""
|
|
197
|
+
found: dict[str, int] = {}
|
|
198
|
+
|
|
199
|
+
def walk(node: Any) -> None:
|
|
200
|
+
if isinstance(node, dict):
|
|
201
|
+
meta = node.get("usage_metadata")
|
|
202
|
+
if isinstance(meta, dict):
|
|
203
|
+
inputs = meta.get("input_token_details") or {}
|
|
204
|
+
outputs = meta.get("output_token_details") or {}
|
|
205
|
+
if "cache_read" in inputs:
|
|
206
|
+
found["cache_read"] = as_int(inputs["cache_read"]) or 0
|
|
207
|
+
writes = [
|
|
208
|
+
as_int(inputs.get(k))
|
|
209
|
+
for k in (
|
|
210
|
+
"cache_creation",
|
|
211
|
+
"ephemeral_5m_input_tokens",
|
|
212
|
+
"ephemeral_1h_input_tokens",
|
|
213
|
+
)
|
|
214
|
+
]
|
|
215
|
+
if any(w is not None for w in writes):
|
|
216
|
+
found["cache_write"] = max(writes[0] or 0, (writes[1] or 0) + (writes[2] or 0))
|
|
217
|
+
if "reasoning" in outputs:
|
|
218
|
+
found["reasoning"] = as_int(outputs["reasoning"]) or 0
|
|
219
|
+
return
|
|
220
|
+
for value in node.values():
|
|
221
|
+
walk(value)
|
|
222
|
+
elif isinstance(node, list):
|
|
223
|
+
for value in node:
|
|
224
|
+
walk(value)
|
|
225
|
+
|
|
226
|
+
walk(maybe_json(raw))
|
|
227
|
+
return found
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def _reasoning_from_raw_output(raw: Any) -> list[str]:
|
|
231
|
+
"""Thinking text from the raw model output.
|
|
232
|
+
|
|
233
|
+
The flattened llm.output_messages attributes drop thinking blocks, but the
|
|
234
|
+
instrumentor also records the raw output (output.value), where the model's
|
|
235
|
+
content blocks are kept, e.g. LangChain's {"type": "thinking", "thinking": ...}.
|
|
236
|
+
We look for such blocks anywhere in it rather than depend on one exact layout.
|
|
237
|
+
"""
|
|
238
|
+
found: list[str] = []
|
|
239
|
+
|
|
240
|
+
def walk(node: Any) -> None:
|
|
241
|
+
if isinstance(node, dict):
|
|
242
|
+
kind = node.get("type")
|
|
243
|
+
if kind in ("thinking", "reasoning"):
|
|
244
|
+
text = node.get("thinking") or node.get("reasoning") or node.get("text")
|
|
245
|
+
if isinstance(text, str) and text.strip():
|
|
246
|
+
found.append(text)
|
|
247
|
+
return
|
|
248
|
+
for value in node.values():
|
|
249
|
+
walk(value)
|
|
250
|
+
elif isinstance(node, list):
|
|
251
|
+
for value in node:
|
|
252
|
+
walk(value)
|
|
253
|
+
|
|
254
|
+
walk(maybe_json(raw))
|
|
255
|
+
# A generation can repeat the same message in several places; keep each once.
|
|
256
|
+
return list(dict.fromkeys(found))
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
# --- flattened messages ------------------------------------------------------------
|
|
260
|
+
# OpenInference flattens lists into indexed keys, for example:
|
|
261
|
+
# llm.input_messages.2.message.role = assistant
|
|
262
|
+
# llm.input_messages.2.message.contents.0.message_content.text = ...
|
|
263
|
+
# llm.input_messages.2.message.tool_calls.0.tool_call.function.name = weekday
|
|
264
|
+
# llm.input_messages.3.message.tool_call_id = toolu_...
|
|
265
|
+
|
|
266
|
+
_INDEXED = re.compile(r"^(\d+)\.(.+)$")
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def _group_by_index(flat: dict[str, Any], prefix: str) -> list[dict[str, Any]]:
|
|
270
|
+
"""{prefix}N.rest = v -> [ {rest: v}, ... ] ordered by N."""
|
|
271
|
+
groups: dict[int, dict[str, Any]] = {}
|
|
272
|
+
for key, value in flat.items():
|
|
273
|
+
if not key.startswith(prefix):
|
|
274
|
+
continue
|
|
275
|
+
match = _INDEXED.match(key[len(prefix) :])
|
|
276
|
+
if match:
|
|
277
|
+
groups.setdefault(int(match.group(1)), {})[match.group(2)] = value
|
|
278
|
+
return [groups[i] for i in sorted(groups)]
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def _messages(attrs: dict[str, Any], prefix: str) -> list[Message]:
|
|
282
|
+
messages = []
|
|
283
|
+
for m in _group_by_index(attrs, prefix):
|
|
284
|
+
parts: list[MessagePart] = []
|
|
285
|
+
if m.get("message.content"):
|
|
286
|
+
text = str(m["message.content"])
|
|
287
|
+
# A tool result: role "tool" (OpenAI), or any message answering a tool
|
|
288
|
+
# call (Anthropic sends results in a "user" message).
|
|
289
|
+
if m.get("message.role") == "tool" or m.get("message.tool_call_id"):
|
|
290
|
+
parts.append(
|
|
291
|
+
MessagePart(
|
|
292
|
+
type="tool_result",
|
|
293
|
+
id=m.get("message.tool_call_id"),
|
|
294
|
+
result=maybe_json(text),
|
|
295
|
+
)
|
|
296
|
+
)
|
|
297
|
+
else:
|
|
298
|
+
parts.append(MessagePart(type="text", text=text))
|
|
299
|
+
for c in _group_by_index(m, "message.contents."):
|
|
300
|
+
kind = c.get("message_content.type", "text")
|
|
301
|
+
if kind == "text":
|
|
302
|
+
parts.append(MessagePart(type="text", text=str(c.get("message_content.text", ""))))
|
|
303
|
+
elif kind == "tool_use":
|
|
304
|
+
continue # the same call is also under message.tool_calls
|
|
305
|
+
else:
|
|
306
|
+
parts.append(MessagePart(type="other", name=c.get("message_content.type")))
|
|
307
|
+
for tc in _group_by_index(m, "message.tool_calls."):
|
|
308
|
+
parts.append(
|
|
309
|
+
MessagePart(
|
|
310
|
+
type="tool_call",
|
|
311
|
+
id=tc.get("tool_call.id"),
|
|
312
|
+
name=tc.get("tool_call.function.name"),
|
|
313
|
+
arguments=maybe_json(tc.get("tool_call.function.arguments")),
|
|
314
|
+
)
|
|
315
|
+
)
|
|
316
|
+
messages.append(Message(role=str(m.get("message.role", "unknown")), parts=parts))
|
|
317
|
+
return messages
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
def _add_raw_tool_results(inputs: list[Message], raw: Any) -> None:
|
|
321
|
+
"""Tool results from the raw request (input.value), which the flattened messages
|
|
322
|
+
can lose: OpenInference's Anthropic instrumentation keeps only one tool result per
|
|
323
|
+
message, and drops Anthropic's is_error flag. Results already present get the
|
|
324
|
+
flag; missing ones are added to the message holding the other results."""
|
|
325
|
+
found: dict[str, tuple[Any, bool | None]] = {}
|
|
326
|
+
|
|
327
|
+
def walk(node: Any) -> None:
|
|
328
|
+
if isinstance(node, dict):
|
|
329
|
+
if node.get("type") == "tool_result" and node.get("tool_use_id"):
|
|
330
|
+
flag = node.get("is_error")
|
|
331
|
+
found[str(node["tool_use_id"])] = (
|
|
332
|
+
_tool_result_content(node.get("content")),
|
|
333
|
+
flag if isinstance(flag, bool) else None,
|
|
334
|
+
)
|
|
335
|
+
return
|
|
336
|
+
for value in node.values():
|
|
337
|
+
walk(value)
|
|
338
|
+
elif isinstance(node, list):
|
|
339
|
+
for value in node:
|
|
340
|
+
walk(value)
|
|
341
|
+
|
|
342
|
+
walk(maybe_json(raw))
|
|
343
|
+
if not found:
|
|
344
|
+
return
|
|
345
|
+
present = {p.id: p for m in inputs for p in m.parts if p.type == "tool_result" and p.id}
|
|
346
|
+
for call_id, (content, flag) in found.items():
|
|
347
|
+
if call_id in present:
|
|
348
|
+
present[call_id].is_error = flag
|
|
349
|
+
continue
|
|
350
|
+
# The message answering the previous calls, or a new one at the end.
|
|
351
|
+
holder = next(
|
|
352
|
+
(m for m in reversed(inputs) if any(p.type == "tool_result" for p in m.parts)), None
|
|
353
|
+
)
|
|
354
|
+
if holder is None:
|
|
355
|
+
holder = Message(role="user", parts=[])
|
|
356
|
+
inputs.append(holder)
|
|
357
|
+
holder.parts.append(
|
|
358
|
+
MessagePart(type="tool_result", id=call_id, result=content, is_error=flag)
|
|
359
|
+
)
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def _tool_result_content(content: Any) -> Any:
|
|
363
|
+
"""Anthropic's tool_result content: a string, or text blocks."""
|
|
364
|
+
if isinstance(content, list):
|
|
365
|
+
texts = [c.get("text") for c in content if isinstance(c, dict) and c.get("text")]
|
|
366
|
+
content = "\n".join(texts) if texts else content
|
|
367
|
+
return maybe_json(content)
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
"""Tool calls rebuilt from the conversation, when no span records them.
|
|
2
|
+
|
|
3
|
+
Instrumenting a model SDK (OpenAI, Anthropic, through OpenInference, OpenLLMetry or
|
|
4
|
+
OpenTelemetry's own packages) records model calls only: the tools are the user's
|
|
5
|
+
own functions, and nothing traces them. The conversation still says what happened.
|
|
6
|
+
A model call's output asks for tools (`tool_call` parts: id, name, arguments), and
|
|
7
|
+
the next model call's input carries their results (`tool_result` parts, same id).
|
|
8
|
+
|
|
9
|
+
So when a run has no tool call spans at all, each requested tool becomes a tool
|
|
10
|
+
call step, marked `synthetic`:
|
|
11
|
+
- name and arguments: from the request;
|
|
12
|
+
- result: from the next model call in the same scope that answers it;
|
|
13
|
+
- start: when the requesting call ended; end: when the answering call started (the
|
|
14
|
+
tool ran somewhere in between; the exact time isn't known). With no answer in
|
|
15
|
+
the run, it lasts no time and has no result;
|
|
16
|
+
- failed: only when the result is flagged as an error (Anthropic's `is_error`).
|
|
17
|
+
The OpenAI API has no such flag, and errors are never guessed from text
|
|
18
|
+
(DECISIONS D44), so there it stays "ok".
|
|
19
|
+
|
|
20
|
+
Runs with any tool call span are left alone: their tools are recorded for real
|
|
21
|
+
(by a framework, or loopview_sdk.tool). A request with no span there is deliberate,
|
|
22
|
+
not lost: structured output, for instance, is a tool call that is never run (the
|
|
23
|
+
flagship's `Plan`). So decorate all of a loop's tools with loopview_sdk.tool, or none.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from loopview.normalize.schema import MessagePart, Step, ToolCall
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def derive_tool_calls(steps: list[Step]) -> list[Step]:
|
|
30
|
+
if any(s.kind == "tool_call" and not s.synthetic for s in steps):
|
|
31
|
+
return steps
|
|
32
|
+
by_scope: dict[str | None, list[Step]] = {}
|
|
33
|
+
for s in steps:
|
|
34
|
+
if s.kind == "model_call" and not s.hidden and s.model is not None:
|
|
35
|
+
by_scope.setdefault(s.scope_id, []).append(s)
|
|
36
|
+
|
|
37
|
+
derived: list[Step] = []
|
|
38
|
+
for calls in by_scope.values():
|
|
39
|
+
calls.sort(key=lambda s: (s.start_ns, s.id))
|
|
40
|
+
for i, call in enumerate(calls):
|
|
41
|
+
requests = [p for m in call.model.output for p in m.parts if p.type == "tool_call"] # type: ignore[union-attr]
|
|
42
|
+
if not requests or call.end_ns is None:
|
|
43
|
+
continue
|
|
44
|
+
later = calls[i + 1 :]
|
|
45
|
+
for n, request in enumerate(requests):
|
|
46
|
+
answer, result = _answer(request, later)
|
|
47
|
+
end = answer.start_ns if answer is not None else call.end_ns
|
|
48
|
+
derived.append(
|
|
49
|
+
Step(
|
|
50
|
+
id=f"{call.id}~tool{n}",
|
|
51
|
+
run_id=call.run_id,
|
|
52
|
+
parent_id=call.scope_id,
|
|
53
|
+
scope_id=call.scope_id,
|
|
54
|
+
kind="tool_call",
|
|
55
|
+
type_label="tool",
|
|
56
|
+
name=request.name or "tool",
|
|
57
|
+
key="", # set when keys are built (loop_nodes._rekey)
|
|
58
|
+
status="error" if result is not None and result.is_error else "ok",
|
|
59
|
+
start_ns=call.end_ns,
|
|
60
|
+
end_ns=max(end, call.end_ns),
|
|
61
|
+
error=_error_text(result),
|
|
62
|
+
tool=ToolCall(
|
|
63
|
+
name=request.name or "tool",
|
|
64
|
+
call_id=request.id,
|
|
65
|
+
arguments=request.arguments,
|
|
66
|
+
result=result.result if result is not None else None,
|
|
67
|
+
),
|
|
68
|
+
convention=call.convention,
|
|
69
|
+
synthetic=True,
|
|
70
|
+
)
|
|
71
|
+
)
|
|
72
|
+
if not derived:
|
|
73
|
+
return steps
|
|
74
|
+
return sorted(steps + derived, key=lambda s: (s.start_ns, s.id))
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _answer(request: MessagePart, later: list[Step]) -> tuple[Step | None, MessagePart | None]:
|
|
78
|
+
"""The first later model call whose input carries this request's result."""
|
|
79
|
+
for call in later:
|
|
80
|
+
for message in call.model.input: # type: ignore[union-attr]
|
|
81
|
+
for part in message.parts:
|
|
82
|
+
if part.type == "tool_result" and request.id and part.id == request.id:
|
|
83
|
+
return call, part
|
|
84
|
+
return None, None
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _error_text(result: MessagePart | None) -> str | None:
|
|
88
|
+
if result is None or not result.is_error:
|
|
89
|
+
return None
|
|
90
|
+
return result.result if isinstance(result.result, str) else "tool error"
|
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
"""Give a flat agent loop the shape of a graph: a `model` node and a `tools` node.
|
|
2
|
+
|
|
3
|
+
LangGraph records a span per graph node, so its agents arrive as `model` and
|
|
4
|
+
`tools` nodes with the calls inside them, and the graph shows the loop. The GenAI
|
|
5
|
+
conventions (Pydantic AI, the Anthropic and OpenAI SDKs, hand instrumented code)
|
|
6
|
+
record only `invoke_agent` with `chat` and `execute_tool` spans directly under it:
|
|
7
|
+
no span says "this is the model step" or "this is the tools step". Drawn as is,
|
|
8
|
+
such an agent is one card holding every call.
|
|
9
|
+
|
|
10
|
+
For an agent with tool calls directly inside it, this pass adds the steps the
|
|
11
|
+
convention leaves out, marked `synthetic`:
|
|
12
|
+
- one `model` step per model call, wrapping it;
|
|
13
|
+
- one `tools` step per turn, wrapping the tool calls that model call asked for.
|
|
14
|
+
A tool call belongs to the agent's latest model call that had ended when the
|
|
15
|
+
tool started (compared with ends, so tied timestamps work, DECISIONS D17).
|
|
16
|
+
|
|
17
|
+
The usual transition rule then draws model -> tools -> model, with a loop counter.
|
|
18
|
+
A flow step started by a tool (a sub-agent run by `gather_facts`) moves inside
|
|
19
|
+
that turn's `tools` step, which then becomes a group. Agents without direct tool
|
|
20
|
+
calls are left alone: with only model calls there is no loop to show.
|
|
21
|
+
|
|
22
|
+
Two more cases from hand-written loops:
|
|
23
|
+
- calls with no flow step around them at all (a model SDK instrumented, nothing
|
|
24
|
+
wrapping the loop) get a synthetic agent, named after the service, so the run
|
|
25
|
+
has something to draw;
|
|
26
|
+
- a plain step that isn't an agent (a generic span such as `main`, or a chain)
|
|
27
|
+
holding both model and tool calls directly is treated as the agent of its loop.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from loopview.normalize.schema import FLOW_KINDS, Step, StepStatus
|
|
31
|
+
|
|
32
|
+
CALL_KINDS = ("model_call", "tool_call")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def add_loop_nodes(steps: list[Step], root_name: str = "agent") -> list[Step]:
|
|
36
|
+
steps = _wrap_unowned_calls(steps, root_name)
|
|
37
|
+
by_id = {s.id: s for s in steps}
|
|
38
|
+
direct: dict[str, list[Step]] = {} # owner id -> its visible direct calls
|
|
39
|
+
for s in steps:
|
|
40
|
+
owner = by_id.get(s.scope_id) if s.scope_id else None
|
|
41
|
+
if owner and not s.hidden and s.kind in CALL_KINDS:
|
|
42
|
+
direct.setdefault(owner.id, []).append(s)
|
|
43
|
+
for owner_id in list(direct):
|
|
44
|
+
owner = by_id[owner_id]
|
|
45
|
+
models = sum(1 for c in direct[owner_id] if c.kind == "model_call")
|
|
46
|
+
tools = len(direct[owner_id]) - models
|
|
47
|
+
# An agent's loop, or another step running a whole loop by itself: several
|
|
48
|
+
# model calls and tools. One model call and its tools is a single turn
|
|
49
|
+
# (the OpenAI Agents SDK's `turn` nodes), already drawn as a step.
|
|
50
|
+
if owner.kind != "agent" and not (models >= 2 and tools >= 1):
|
|
51
|
+
del direct[owner_id]
|
|
52
|
+
|
|
53
|
+
added: list[Step] = []
|
|
54
|
+
new_scope: dict[str, str] = {} # call id -> the synthetic step that now holds it
|
|
55
|
+
for agent_id, calls in direct.items():
|
|
56
|
+
if not any(c.kind == "tool_call" for c in calls):
|
|
57
|
+
continue
|
|
58
|
+
agent = by_id[agent_id]
|
|
59
|
+
if agent.kind != "agent":
|
|
60
|
+
agent.kind = "agent" # it holds its loop's nodes now: a group
|
|
61
|
+
models = sorted((c for c in calls if c.kind == "model_call"), key=_order)
|
|
62
|
+
for m in models:
|
|
63
|
+
step = _synthetic(agent, f"{m.id}~model", "model", [m])
|
|
64
|
+
added.append(step)
|
|
65
|
+
new_scope[m.id] = step.id
|
|
66
|
+
turns: dict[int, list[Step]] = {}
|
|
67
|
+
for t in sorted((c for c in calls if c.kind == "tool_call"), key=_order):
|
|
68
|
+
turns.setdefault(_turn_of(t, models), []).append(t)
|
|
69
|
+
for index, tools in turns.items():
|
|
70
|
+
step_id = f"{models[index].id}~tools" if index >= 0 else f"{agent.id}~tools"
|
|
71
|
+
step = _synthetic(agent, step_id, "tools", tools)
|
|
72
|
+
added.append(step)
|
|
73
|
+
for t in tools:
|
|
74
|
+
new_scope[t.id] = step.id
|
|
75
|
+
|
|
76
|
+
if not added:
|
|
77
|
+
if any(not s.key for s in steps):
|
|
78
|
+
_rekey(steps, by_id) # tool calls rebuilt from messages have no key yet
|
|
79
|
+
return steps
|
|
80
|
+
all_steps = steps + added
|
|
81
|
+
by_id.update({s.id: s for s in added})
|
|
82
|
+
|
|
83
|
+
# Move the calls, and everything that happened inside them, into their new step.
|
|
84
|
+
for s in steps:
|
|
85
|
+
if s.id in new_scope:
|
|
86
|
+
s.parent_id = s.scope_id = new_scope[s.id]
|
|
87
|
+
continue
|
|
88
|
+
inside = _enclosing_call(s, by_id, new_scope)
|
|
89
|
+
if inside is not None:
|
|
90
|
+
s.scope_id = new_scope[inside]
|
|
91
|
+
|
|
92
|
+
# A tools step that now holds flow steps (sub-agents) is a group.
|
|
93
|
+
for s in all_steps:
|
|
94
|
+
if s.kind in FLOW_KINDS and not s.hidden and s.scope_id:
|
|
95
|
+
holder = by_id[s.scope_id]
|
|
96
|
+
if holder.synthetic and holder.kind == "node":
|
|
97
|
+
holder.kind = "agent"
|
|
98
|
+
|
|
99
|
+
_rekey(all_steps, by_id)
|
|
100
|
+
all_steps.sort(key=_order)
|
|
101
|
+
return all_steps
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _wrap_unowned_calls(steps: list[Step], root_name: str) -> list[Step]:
|
|
105
|
+
"""Model and tool calls with no flow step around them get a synthetic agent."""
|
|
106
|
+
unowned = [s for s in steps if s.kind in CALL_KINDS and not s.hidden and s.scope_id is None]
|
|
107
|
+
if not unowned:
|
|
108
|
+
return steps
|
|
109
|
+
first = min(unowned, key=_order)
|
|
110
|
+
running = any(s.end_ns is None for s in unowned)
|
|
111
|
+
root = Step(
|
|
112
|
+
id=f"{first.run_id}~agent",
|
|
113
|
+
run_id=first.run_id,
|
|
114
|
+
parent_id=None,
|
|
115
|
+
scope_id=None,
|
|
116
|
+
kind="agent",
|
|
117
|
+
type_label="agent",
|
|
118
|
+
name=root_name,
|
|
119
|
+
key="",
|
|
120
|
+
status="running" if running else "ok",
|
|
121
|
+
start_ns=first.start_ns,
|
|
122
|
+
end_ns=None if running else max(s.end_ns or 0 for s in unowned),
|
|
123
|
+
convention=first.convention,
|
|
124
|
+
synthetic=True,
|
|
125
|
+
)
|
|
126
|
+
for s in unowned:
|
|
127
|
+
s.scope_id = root.id
|
|
128
|
+
if s.parent_id is None:
|
|
129
|
+
s.parent_id = root.id
|
|
130
|
+
return [root, *steps]
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _order(s: Step) -> tuple[int, str]:
|
|
134
|
+
return (s.start_ns, s.id)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _turn_of(call: Step, models: list[Step]) -> int:
|
|
138
|
+
turn = -1
|
|
139
|
+
for i, m in enumerate(models):
|
|
140
|
+
if m.end_ns is not None and m.end_ns <= call.start_ns:
|
|
141
|
+
turn = i
|
|
142
|
+
return turn
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _synthetic(agent: Step, step_id: str, name: str, inside: list[Step]) -> Step:
|
|
146
|
+
running = any(c.end_ns is None for c in inside)
|
|
147
|
+
status: StepStatus = "running" if running else "ok"
|
|
148
|
+
if name == "model" and not running:
|
|
149
|
+
status = inside[0].status
|
|
150
|
+
return Step(
|
|
151
|
+
id=step_id,
|
|
152
|
+
run_id=agent.run_id,
|
|
153
|
+
parent_id=agent.id,
|
|
154
|
+
scope_id=agent.id,
|
|
155
|
+
kind="node",
|
|
156
|
+
type_label=name,
|
|
157
|
+
name=name,
|
|
158
|
+
key="", # set by _rekey
|
|
159
|
+
status=status,
|
|
160
|
+
start_ns=min(c.start_ns for c in inside),
|
|
161
|
+
end_ns=None if running else max(c.end_ns or 0 for c in inside),
|
|
162
|
+
convention=agent.convention,
|
|
163
|
+
synthetic=True,
|
|
164
|
+
)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _enclosing_call(s: Step, by_id: dict[str, Step], moved: dict[str, str]) -> str | None:
|
|
168
|
+
"""The moved call that `s` happened inside, if any (walking visible parents)."""
|
|
169
|
+
parent_id = s.parent_id
|
|
170
|
+
while parent_id is not None and parent_id in by_id:
|
|
171
|
+
if parent_id in moved:
|
|
172
|
+
return parent_id
|
|
173
|
+
parent = by_id[parent_id]
|
|
174
|
+
if parent.kind in FLOW_KINDS:
|
|
175
|
+
return None # another flow step owns it
|
|
176
|
+
parent_id = parent.parent_id
|
|
177
|
+
return None
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _rekey(steps: list[Step], by_id: dict[str, Step]) -> None:
|
|
181
|
+
"""Keys are the scope path plus the name, as the normalizer builds them; scopes
|
|
182
|
+
changed, so build them again."""
|
|
183
|
+
keys: dict[str, str] = {}
|
|
184
|
+
|
|
185
|
+
def key_of(step: Step) -> str:
|
|
186
|
+
if step.id not in keys:
|
|
187
|
+
name = step.name if step.kind in FLOW_KINDS else f"{step.kind}:{step.name}"
|
|
188
|
+
scope = by_id.get(step.scope_id) if step.scope_id else None
|
|
189
|
+
keys[step.id] = f"{key_of(scope)}/{name}" if scope else name
|
|
190
|
+
return keys[step.id]
|
|
191
|
+
|
|
192
|
+
for s in steps:
|
|
193
|
+
s.key = key_of(s)
|