loopview 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. loopview/__init__.py +3 -0
  2. loopview/app.py +252 -0
  3. loopview/cli.py +128 -0
  4. loopview/cost/__init__.py +1 -0
  5. loopview/cost/pricing.json +62 -0
  6. loopview/cost/pricing.py +74 -0
  7. loopview/cost/split.py +228 -0
  8. loopview/demo_data/flagship.otlp.jsonl +13 -0
  9. loopview/devtools/__init__.py +1 -0
  10. loopview/devtools/dump_normalized.py +211 -0
  11. loopview/ingest/__init__.py +1 -0
  12. loopview/ingest/otlp.py +305 -0
  13. loopview/ingest/raw.py +47 -0
  14. loopview/live.py +154 -0
  15. loopview/normalize/__init__.py +1 -0
  16. loopview/normalize/adapters/__init__.py +14 -0
  17. loopview/normalize/adapters/base.py +87 -0
  18. loopview/normalize/adapters/gen_ai.py +252 -0
  19. loopview/normalize/adapters/generic.py +22 -0
  20. loopview/normalize/adapters/openinference.py +367 -0
  21. loopview/normalize/derived_tools.py +90 -0
  22. loopview/normalize/loop_nodes.py +193 -0
  23. loopview/normalize/normalizer.py +326 -0
  24. loopview/normalize/schema.py +161 -0
  25. loopview/normalize/transitions.py +106 -0
  26. loopview/py.typed +0 -0
  27. loopview/static/assets/elk-worker.min-DfmSo98M.js +22 -0
  28. loopview/static/assets/index-CEGppBgk.css +1 -0
  29. loopview/static/assets/index-DzC96LiQ.js +21 -0
  30. loopview/static/assets/inter-cyrillic-ext-wght-normal-BOeWTOD4.woff2 +0 -0
  31. loopview/static/assets/inter-cyrillic-wght-normal-DqGufNeO.woff2 +0 -0
  32. loopview/static/assets/inter-greek-ext-wght-normal-DlzME5K_.woff2 +0 -0
  33. loopview/static/assets/inter-greek-wght-normal-CkhJZR-_.woff2 +0 -0
  34. loopview/static/assets/inter-latin-ext-wght-normal-DO1Apj_S.woff2 +0 -0
  35. loopview/static/assets/inter-latin-wght-normal-Dx4kXJAl.woff2 +0 -0
  36. loopview/static/assets/inter-vietnamese-wght-normal-CBcvBZtf.woff2 +0 -0
  37. loopview/static/assets/jetbrains-mono-cyrillic-wght-normal-D73BlboJ.woff2 +0 -0
  38. loopview/static/assets/jetbrains-mono-greek-wght-normal-Bw9x6K1M.woff2 +0 -0
  39. loopview/static/assets/jetbrains-mono-latin-ext-wght-normal-DBQx-q_a.woff2 +0 -0
  40. loopview/static/assets/jetbrains-mono-latin-wght-normal-B9CIFXIH.woff2 +0 -0
  41. loopview/static/assets/jetbrains-mono-vietnamese-wght-normal-Bt-aOZkq.woff2 +0 -0
  42. loopview/static/index.html +13 -0
  43. loopview/store/__init__.py +1 -0
  44. loopview/store/capture.py +112 -0
  45. loopview/store/memory.py +165 -0
  46. loopview/tools/__init__.py +1 -0
  47. loopview/tools/report.py +347 -0
  48. loopview-0.1.0.dist-info/METADATA +72 -0
  49. loopview-0.1.0.dist-info/RECORD +51 -0
  50. loopview-0.1.0.dist-info/WHEEL +4 -0
  51. loopview-0.1.0.dist-info/entry_points.txt +3 -0
@@ -0,0 +1,367 @@
1
+ """Adapter for OpenInference (Arize) semantic conventions.
2
+
3
+ Written against the spec in Arize-ai/openinference (spec/semantic_conventions.md)
4
+ as emitted by openinference-instrumentation-langchain 0.1.76 and
5
+ openinference-semantic-conventions 0.1.39. The spec itself has no version number.
6
+
7
+ A span is OpenInference if it has openinference.span.kind.
8
+
9
+ LangGraph: the LangChain instrumentor copies LangGraph's run metadata into the
10
+ `metadata` attribute. That is how we tell a real graph node (langgraph_node ==
11
+ span name) from plumbing inside a node, such as a routing function or an output
12
+ parser (langgraph_node != span name), which we keep but hide.
13
+ """
14
+
15
+ import re
16
+ from typing import Any
17
+
18
+ from loopview.ingest.raw import RawSpan
19
+ from loopview.normalize.adapters.base import (
20
+ Classification,
21
+ ParentHint,
22
+ as_int,
23
+ maybe_json,
24
+ )
25
+ from loopview.normalize.schema import Message, MessagePart, ModelCall, ToolCall, Usage
26
+
27
+ TOOL_LIKE_KINDS = {
28
+ "TOOL": "tool",
29
+ "RETRIEVER": "retriever",
30
+ "EMBEDDING": "embedding",
31
+ "RERANKER": "reranker",
32
+ }
33
+
34
+
35
+ class OpenInferenceAdapter:
36
+ name = "openinference"
37
+
38
+ def matches(self, span: RawSpan) -> bool:
39
+ return "openinference.span.kind" in span.attributes
40
+
41
+ def classify(self, span: RawSpan) -> Classification:
42
+ attrs = span.attributes
43
+ kind = str(attrs["openinference.span.kind"]).upper()
44
+ io = {
45
+ "input": maybe_json(attrs.get("input.value")),
46
+ "output": maybe_json(attrs.get("output.value")),
47
+ }
48
+
49
+ if kind == "LLM":
50
+ model = attrs.get("llm.model_name") or span.name
51
+ return Classification("model_call", str(model), "llm", model=_model_call(attrs))
52
+ if kind in TOOL_LIKE_KINDS:
53
+ name = attrs.get("tool.name") or span.name
54
+ return Classification(
55
+ "tool_call",
56
+ str(name),
57
+ TOOL_LIKE_KINDS[kind],
58
+ tool=ToolCall(
59
+ name=str(name),
60
+ call_id=attrs.get("tool_call.id"),
61
+ arguments=io["input"],
62
+ result=_unwrap_tool_output(io["output"]),
63
+ ),
64
+ )
65
+ if kind == "PROMPT":
66
+ return Classification("node", span.name, "prompt", hidden=True, **io)
67
+ if kind in ("GUARDRAIL", "EVALUATOR"):
68
+ return Classification("node", span.name, kind.lower(), **io)
69
+
70
+ # CHAIN and AGENT: graphs, graph nodes, chains, agents.
71
+ metadata = _metadata(attrs)
72
+ graph_node = metadata.get("langgraph_node")
73
+ if graph_node is None and metadata.get("ls_integration") == "langgraph":
74
+ return Classification("agent", span.name, "graph", **io) # a compiled graph
75
+ if graph_node is not None and graph_node != span.name:
76
+ return Classification("node", span.name, "internal", hidden=True, **io)
77
+ if graph_node == span.name:
78
+ # A real node. If a subgraph runs inside it, the subgraph's own span has
79
+ # the same name and sits directly under it: collapse the two.
80
+ return Classification("node", span.name, "node", collapse_into_parent=True, **io)
81
+ if kind == "AGENT":
82
+ # graph.node.id is the agent's own name where the span name is a method
83
+ # (CrewAI: "Weather assistant._execute_core").
84
+ name = attrs.get("agent.name") or attrs.get("graph.node.id") or _readable(span.name)
85
+ return Classification("agent", str(name), "agent", **io)
86
+ # A chain directly inside a span of the same name is the same step recorded
87
+ # twice (the OpenAI Agents SDK wraps its "Agent workflow" AGENT in a CHAIN).
88
+ return Classification(
89
+ "node", _readable(span.name), "chain", collapse_into_parent=True, **io
90
+ )
91
+
92
+ def infer_parent(self, span: RawSpan) -> ParentHint | None:
93
+ metadata = _metadata(span.attributes)
94
+ graph_node = metadata.get("langgraph_node")
95
+ if graph_node is None:
96
+ return None
97
+ if graph_node != span.name:
98
+ return ParentHint(str(graph_node), "node", "node") # plumbing inside a node
99
+ # A node inside a subgraph: the namespace is "outer:<id>|inner:<id>".
100
+ namespace = str(metadata.get("langgraph_checkpoint_ns", "")).split("|")
101
+ if len(namespace) > 1:
102
+ return ParentHint(namespace[-2].split(":")[0], "agent", "graph")
103
+ return None
104
+
105
+
106
+ _UUID_IN_NAME = re.compile(
107
+ r"[_-]?[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}"
108
+ )
109
+
110
+
111
+ def _readable(name: str) -> str:
112
+ """A span name without the run's UUID in it (CrewAI: "Crew_<uuid>.kickoff")."""
113
+ return _UUID_IN_NAME.sub("", name) or name
114
+
115
+
116
+ def _metadata(attrs: dict[str, Any]) -> dict[str, Any]:
117
+ value = maybe_json(attrs.get("metadata"))
118
+ return value if isinstance(value, dict) else {}
119
+
120
+
121
+ def _unwrap_tool_output(value: Any) -> Any:
122
+ """LangChain serialises a tool result as a ToolMessage; show just its content."""
123
+ if (
124
+ isinstance(value, dict)
125
+ and value.get("type") == "tool"
126
+ and isinstance(value.get("data"), dict)
127
+ ):
128
+ return maybe_json(value["data"].get("content"))
129
+ return value
130
+
131
+
132
+ def _model_call(attrs: dict[str, Any]) -> ModelCall:
133
+ output = _messages(attrs, "llm.output_messages.")
134
+ reasoning = _reasoning_from_raw_output(attrs.get("output.value"))
135
+ if reasoning:
136
+ if not output:
137
+ output = [Message(role="assistant", parts=[])]
138
+ output[0].parts[:0] = [MessagePart(type="reasoning", text=r) for r in reasoning]
139
+ inputs = _messages(attrs, "llm.input_messages.")
140
+ _add_raw_tool_results(inputs, attrs.get("input.value"))
141
+ return ModelCall(
142
+ provider=attrs.get("llm.provider") or attrs.get("llm.system"),
143
+ model=attrs.get("llm.model_name"),
144
+ input=inputs,
145
+ output=output,
146
+ tool_definitions=_tool_definitions(attrs),
147
+ usage=_usage(attrs),
148
+ )
149
+
150
+
151
+ def _tool_definitions(attrs: dict[str, Any]) -> list[Any]:
152
+ """llm.tools.N.tool.json_schema, in order."""
153
+ return [
154
+ maybe_json(t.get("tool.json_schema"))
155
+ for t in _group_by_index(attrs, "llm.tools.")
156
+ if t.get("tool.json_schema") is not None
157
+ ]
158
+
159
+
160
+ def _usage(attrs: dict[str, Any]) -> Usage | None:
161
+ """Token counts from the spec's attributes, completed from the raw output.
162
+
163
+ The instrumentor leaves out cache attributes that are zero, and doesn't copy
164
+ reasoning (thinking) tokens into an attribute at all. Both are in the raw
165
+ output it also records, as LangChain's usage_metadata."""
166
+ raw = _raw_usage(attrs.get("output.value"))
167
+ usage = Usage(
168
+ input_tokens=as_int(attrs.get("llm.token_count.prompt")),
169
+ output_tokens=as_int(attrs.get("llm.token_count.completion")),
170
+ cache_read_tokens=_first(
171
+ attrs.get("llm.token_count.prompt_details.cache_read"), raw.get("cache_read")
172
+ ),
173
+ cache_write_tokens=_first(
174
+ attrs.get("llm.token_count.prompt_details.cache_write"), raw.get("cache_write")
175
+ ),
176
+ reasoning_tokens=_first(
177
+ attrs.get("llm.token_count.completion_details.reasoning"), raw.get("reasoning")
178
+ ),
179
+ )
180
+ return usage if usage.model_dump(exclude_none=True) else None
181
+
182
+
183
+ def _first(*values: Any) -> int | None:
184
+ for value in values:
185
+ if as_int(value) is not None:
186
+ return as_int(value)
187
+ return None
188
+
189
+
190
+ def _raw_usage(raw: Any) -> dict[str, int]:
191
+ """Cache and reasoning counts from LangChain's usage_metadata in the raw output:
192
+ {"input_token_details": {"cache_read", "cache_creation", "ephemeral_5m_input_tokens",
193
+ "ephemeral_1h_input_tokens"}, "output_token_details": {"reasoning"}}.
194
+
195
+ LangChain's cache_creation can read 0 while the per-lifetime counts
196
+ (ephemeral_5m / ephemeral_1h) hold the cache writes, so the larger wins."""
197
+ found: dict[str, int] = {}
198
+
199
+ def walk(node: Any) -> None:
200
+ if isinstance(node, dict):
201
+ meta = node.get("usage_metadata")
202
+ if isinstance(meta, dict):
203
+ inputs = meta.get("input_token_details") or {}
204
+ outputs = meta.get("output_token_details") or {}
205
+ if "cache_read" in inputs:
206
+ found["cache_read"] = as_int(inputs["cache_read"]) or 0
207
+ writes = [
208
+ as_int(inputs.get(k))
209
+ for k in (
210
+ "cache_creation",
211
+ "ephemeral_5m_input_tokens",
212
+ "ephemeral_1h_input_tokens",
213
+ )
214
+ ]
215
+ if any(w is not None for w in writes):
216
+ found["cache_write"] = max(writes[0] or 0, (writes[1] or 0) + (writes[2] or 0))
217
+ if "reasoning" in outputs:
218
+ found["reasoning"] = as_int(outputs["reasoning"]) or 0
219
+ return
220
+ for value in node.values():
221
+ walk(value)
222
+ elif isinstance(node, list):
223
+ for value in node:
224
+ walk(value)
225
+
226
+ walk(maybe_json(raw))
227
+ return found
228
+
229
+
230
+ def _reasoning_from_raw_output(raw: Any) -> list[str]:
231
+ """Thinking text from the raw model output.
232
+
233
+ The flattened llm.output_messages attributes drop thinking blocks, but the
234
+ instrumentor also records the raw output (output.value), where the model's
235
+ content blocks are kept, e.g. LangChain's {"type": "thinking", "thinking": ...}.
236
+ We look for such blocks anywhere in it rather than depend on one exact layout.
237
+ """
238
+ found: list[str] = []
239
+
240
+ def walk(node: Any) -> None:
241
+ if isinstance(node, dict):
242
+ kind = node.get("type")
243
+ if kind in ("thinking", "reasoning"):
244
+ text = node.get("thinking") or node.get("reasoning") or node.get("text")
245
+ if isinstance(text, str) and text.strip():
246
+ found.append(text)
247
+ return
248
+ for value in node.values():
249
+ walk(value)
250
+ elif isinstance(node, list):
251
+ for value in node:
252
+ walk(value)
253
+
254
+ walk(maybe_json(raw))
255
+ # A generation can repeat the same message in several places; keep each once.
256
+ return list(dict.fromkeys(found))
257
+
258
+
259
+ # --- flattened messages ------------------------------------------------------------
260
+ # OpenInference flattens lists into indexed keys, for example:
261
+ # llm.input_messages.2.message.role = assistant
262
+ # llm.input_messages.2.message.contents.0.message_content.text = ...
263
+ # llm.input_messages.2.message.tool_calls.0.tool_call.function.name = weekday
264
+ # llm.input_messages.3.message.tool_call_id = toolu_...
265
+
266
+ _INDEXED = re.compile(r"^(\d+)\.(.+)$")
267
+
268
+
269
+ def _group_by_index(flat: dict[str, Any], prefix: str) -> list[dict[str, Any]]:
270
+ """{prefix}N.rest = v -> [ {rest: v}, ... ] ordered by N."""
271
+ groups: dict[int, dict[str, Any]] = {}
272
+ for key, value in flat.items():
273
+ if not key.startswith(prefix):
274
+ continue
275
+ match = _INDEXED.match(key[len(prefix) :])
276
+ if match:
277
+ groups.setdefault(int(match.group(1)), {})[match.group(2)] = value
278
+ return [groups[i] for i in sorted(groups)]
279
+
280
+
281
+ def _messages(attrs: dict[str, Any], prefix: str) -> list[Message]:
282
+ messages = []
283
+ for m in _group_by_index(attrs, prefix):
284
+ parts: list[MessagePart] = []
285
+ if m.get("message.content"):
286
+ text = str(m["message.content"])
287
+ # A tool result: role "tool" (OpenAI), or any message answering a tool
288
+ # call (Anthropic sends results in a "user" message).
289
+ if m.get("message.role") == "tool" or m.get("message.tool_call_id"):
290
+ parts.append(
291
+ MessagePart(
292
+ type="tool_result",
293
+ id=m.get("message.tool_call_id"),
294
+ result=maybe_json(text),
295
+ )
296
+ )
297
+ else:
298
+ parts.append(MessagePart(type="text", text=text))
299
+ for c in _group_by_index(m, "message.contents."):
300
+ kind = c.get("message_content.type", "text")
301
+ if kind == "text":
302
+ parts.append(MessagePart(type="text", text=str(c.get("message_content.text", ""))))
303
+ elif kind == "tool_use":
304
+ continue # the same call is also under message.tool_calls
305
+ else:
306
+ parts.append(MessagePart(type="other", name=c.get("message_content.type")))
307
+ for tc in _group_by_index(m, "message.tool_calls."):
308
+ parts.append(
309
+ MessagePart(
310
+ type="tool_call",
311
+ id=tc.get("tool_call.id"),
312
+ name=tc.get("tool_call.function.name"),
313
+ arguments=maybe_json(tc.get("tool_call.function.arguments")),
314
+ )
315
+ )
316
+ messages.append(Message(role=str(m.get("message.role", "unknown")), parts=parts))
317
+ return messages
318
+
319
+
320
+ def _add_raw_tool_results(inputs: list[Message], raw: Any) -> None:
321
+ """Tool results from the raw request (input.value), which the flattened messages
322
+ can lose: OpenInference's Anthropic instrumentation keeps only one tool result per
323
+ message, and drops Anthropic's is_error flag. Results already present get the
324
+ flag; missing ones are added to the message holding the other results."""
325
+ found: dict[str, tuple[Any, bool | None]] = {}
326
+
327
+ def walk(node: Any) -> None:
328
+ if isinstance(node, dict):
329
+ if node.get("type") == "tool_result" and node.get("tool_use_id"):
330
+ flag = node.get("is_error")
331
+ found[str(node["tool_use_id"])] = (
332
+ _tool_result_content(node.get("content")),
333
+ flag if isinstance(flag, bool) else None,
334
+ )
335
+ return
336
+ for value in node.values():
337
+ walk(value)
338
+ elif isinstance(node, list):
339
+ for value in node:
340
+ walk(value)
341
+
342
+ walk(maybe_json(raw))
343
+ if not found:
344
+ return
345
+ present = {p.id: p for m in inputs for p in m.parts if p.type == "tool_result" and p.id}
346
+ for call_id, (content, flag) in found.items():
347
+ if call_id in present:
348
+ present[call_id].is_error = flag
349
+ continue
350
+ # The message answering the previous calls, or a new one at the end.
351
+ holder = next(
352
+ (m for m in reversed(inputs) if any(p.type == "tool_result" for p in m.parts)), None
353
+ )
354
+ if holder is None:
355
+ holder = Message(role="user", parts=[])
356
+ inputs.append(holder)
357
+ holder.parts.append(
358
+ MessagePart(type="tool_result", id=call_id, result=content, is_error=flag)
359
+ )
360
+
361
+
362
+ def _tool_result_content(content: Any) -> Any:
363
+ """Anthropic's tool_result content: a string, or text blocks."""
364
+ if isinstance(content, list):
365
+ texts = [c.get("text") for c in content if isinstance(c, dict) and c.get("text")]
366
+ content = "\n".join(texts) if texts else content
367
+ return maybe_json(content)
@@ -0,0 +1,90 @@
1
+ """Tool calls rebuilt from the conversation, when no span records them.
2
+
3
+ Instrumenting a model SDK (OpenAI, Anthropic, through OpenInference, OpenLLMetry or
4
+ OpenTelemetry's own packages) records model calls only: the tools are the user's
5
+ own functions, and nothing traces them. The conversation still says what happened.
6
+ A model call's output asks for tools (`tool_call` parts: id, name, arguments), and
7
+ the next model call's input carries their results (`tool_result` parts, same id).
8
+
9
+ So when a run has no tool call spans at all, each requested tool becomes a tool
10
+ call step, marked `synthetic`:
11
+ - name and arguments: from the request;
12
+ - result: from the next model call in the same scope that answers it;
13
+ - start: when the requesting call ended; end: when the answering call started (the
14
+ tool ran somewhere in between; the exact time isn't known). With no answer in
15
+ the run, it lasts no time and has no result;
16
+ - failed: only when the result is flagged as an error (Anthropic's `is_error`).
17
+ The OpenAI API has no such flag, and errors are never guessed from text
18
+ (DECISIONS D44), so there it stays "ok".
19
+
20
+ Runs with any tool call span are left alone: their tools are recorded for real
21
+ (by a framework, or loopview_sdk.tool). A request with no span there is deliberate,
22
+ not lost: structured output, for instance, is a tool call that is never run (the
23
+ flagship's `Plan`). So decorate all of a loop's tools with loopview_sdk.tool, or none.
24
+ """
25
+
26
+ from loopview.normalize.schema import MessagePart, Step, ToolCall
27
+
28
+
29
+ def derive_tool_calls(steps: list[Step]) -> list[Step]:
30
+ if any(s.kind == "tool_call" and not s.synthetic for s in steps):
31
+ return steps
32
+ by_scope: dict[str | None, list[Step]] = {}
33
+ for s in steps:
34
+ if s.kind == "model_call" and not s.hidden and s.model is not None:
35
+ by_scope.setdefault(s.scope_id, []).append(s)
36
+
37
+ derived: list[Step] = []
38
+ for calls in by_scope.values():
39
+ calls.sort(key=lambda s: (s.start_ns, s.id))
40
+ for i, call in enumerate(calls):
41
+ requests = [p for m in call.model.output for p in m.parts if p.type == "tool_call"] # type: ignore[union-attr]
42
+ if not requests or call.end_ns is None:
43
+ continue
44
+ later = calls[i + 1 :]
45
+ for n, request in enumerate(requests):
46
+ answer, result = _answer(request, later)
47
+ end = answer.start_ns if answer is not None else call.end_ns
48
+ derived.append(
49
+ Step(
50
+ id=f"{call.id}~tool{n}",
51
+ run_id=call.run_id,
52
+ parent_id=call.scope_id,
53
+ scope_id=call.scope_id,
54
+ kind="tool_call",
55
+ type_label="tool",
56
+ name=request.name or "tool",
57
+ key="", # set when keys are built (loop_nodes._rekey)
58
+ status="error" if result is not None and result.is_error else "ok",
59
+ start_ns=call.end_ns,
60
+ end_ns=max(end, call.end_ns),
61
+ error=_error_text(result),
62
+ tool=ToolCall(
63
+ name=request.name or "tool",
64
+ call_id=request.id,
65
+ arguments=request.arguments,
66
+ result=result.result if result is not None else None,
67
+ ),
68
+ convention=call.convention,
69
+ synthetic=True,
70
+ )
71
+ )
72
+ if not derived:
73
+ return steps
74
+ return sorted(steps + derived, key=lambda s: (s.start_ns, s.id))
75
+
76
+
77
+ def _answer(request: MessagePart, later: list[Step]) -> tuple[Step | None, MessagePart | None]:
78
+ """The first later model call whose input carries this request's result."""
79
+ for call in later:
80
+ for message in call.model.input: # type: ignore[union-attr]
81
+ for part in message.parts:
82
+ if part.type == "tool_result" and request.id and part.id == request.id:
83
+ return call, part
84
+ return None, None
85
+
86
+
87
+ def _error_text(result: MessagePart | None) -> str | None:
88
+ if result is None or not result.is_error:
89
+ return None
90
+ return result.result if isinstance(result.result, str) else "tool error"
@@ -0,0 +1,193 @@
1
+ """Give a flat agent loop the shape of a graph: a `model` node and a `tools` node.
2
+
3
+ LangGraph records a span per graph node, so its agents arrive as `model` and
4
+ `tools` nodes with the calls inside them, and the graph shows the loop. The GenAI
5
+ conventions (Pydantic AI, the Anthropic and OpenAI SDKs, hand instrumented code)
6
+ record only `invoke_agent` with `chat` and `execute_tool` spans directly under it:
7
+ no span says "this is the model step" or "this is the tools step". Drawn as is,
8
+ such an agent is one card holding every call.
9
+
10
+ For an agent with tool calls directly inside it, this pass adds the steps the
11
+ convention leaves out, marked `synthetic`:
12
+ - one `model` step per model call, wrapping it;
13
+ - one `tools` step per turn, wrapping the tool calls that model call asked for.
14
+ A tool call belongs to the agent's latest model call that had ended when the
15
+ tool started (compared with ends, so tied timestamps work, DECISIONS D17).
16
+
17
+ The usual transition rule then draws model -> tools -> model, with a loop counter.
18
+ A flow step started by a tool (a sub-agent run by `gather_facts`) moves inside
19
+ that turn's `tools` step, which then becomes a group. Agents without direct tool
20
+ calls are left alone: with only model calls there is no loop to show.
21
+
22
+ Two more cases from hand-written loops:
23
+ - calls with no flow step around them at all (a model SDK instrumented, nothing
24
+ wrapping the loop) get a synthetic agent, named after the service, so the run
25
+ has something to draw;
26
+ - a plain step that isn't an agent (a generic span such as `main`, or a chain)
27
+ holding both model and tool calls directly is treated as the agent of its loop.
28
+ """
29
+
30
+ from loopview.normalize.schema import FLOW_KINDS, Step, StepStatus
31
+
32
+ CALL_KINDS = ("model_call", "tool_call")
33
+
34
+
35
+ def add_loop_nodes(steps: list[Step], root_name: str = "agent") -> list[Step]:
36
+ steps = _wrap_unowned_calls(steps, root_name)
37
+ by_id = {s.id: s for s in steps}
38
+ direct: dict[str, list[Step]] = {} # owner id -> its visible direct calls
39
+ for s in steps:
40
+ owner = by_id.get(s.scope_id) if s.scope_id else None
41
+ if owner and not s.hidden and s.kind in CALL_KINDS:
42
+ direct.setdefault(owner.id, []).append(s)
43
+ for owner_id in list(direct):
44
+ owner = by_id[owner_id]
45
+ models = sum(1 for c in direct[owner_id] if c.kind == "model_call")
46
+ tools = len(direct[owner_id]) - models
47
+ # An agent's loop, or another step running a whole loop by itself: several
48
+ # model calls and tools. One model call and its tools is a single turn
49
+ # (the OpenAI Agents SDK's `turn` nodes), already drawn as a step.
50
+ if owner.kind != "agent" and not (models >= 2 and tools >= 1):
51
+ del direct[owner_id]
52
+
53
+ added: list[Step] = []
54
+ new_scope: dict[str, str] = {} # call id -> the synthetic step that now holds it
55
+ for agent_id, calls in direct.items():
56
+ if not any(c.kind == "tool_call" for c in calls):
57
+ continue
58
+ agent = by_id[agent_id]
59
+ if agent.kind != "agent":
60
+ agent.kind = "agent" # it holds its loop's nodes now: a group
61
+ models = sorted((c for c in calls if c.kind == "model_call"), key=_order)
62
+ for m in models:
63
+ step = _synthetic(agent, f"{m.id}~model", "model", [m])
64
+ added.append(step)
65
+ new_scope[m.id] = step.id
66
+ turns: dict[int, list[Step]] = {}
67
+ for t in sorted((c for c in calls if c.kind == "tool_call"), key=_order):
68
+ turns.setdefault(_turn_of(t, models), []).append(t)
69
+ for index, tools in turns.items():
70
+ step_id = f"{models[index].id}~tools" if index >= 0 else f"{agent.id}~tools"
71
+ step = _synthetic(agent, step_id, "tools", tools)
72
+ added.append(step)
73
+ for t in tools:
74
+ new_scope[t.id] = step.id
75
+
76
+ if not added:
77
+ if any(not s.key for s in steps):
78
+ _rekey(steps, by_id) # tool calls rebuilt from messages have no key yet
79
+ return steps
80
+ all_steps = steps + added
81
+ by_id.update({s.id: s for s in added})
82
+
83
+ # Move the calls, and everything that happened inside them, into their new step.
84
+ for s in steps:
85
+ if s.id in new_scope:
86
+ s.parent_id = s.scope_id = new_scope[s.id]
87
+ continue
88
+ inside = _enclosing_call(s, by_id, new_scope)
89
+ if inside is not None:
90
+ s.scope_id = new_scope[inside]
91
+
92
+ # A tools step that now holds flow steps (sub-agents) is a group.
93
+ for s in all_steps:
94
+ if s.kind in FLOW_KINDS and not s.hidden and s.scope_id:
95
+ holder = by_id[s.scope_id]
96
+ if holder.synthetic and holder.kind == "node":
97
+ holder.kind = "agent"
98
+
99
+ _rekey(all_steps, by_id)
100
+ all_steps.sort(key=_order)
101
+ return all_steps
102
+
103
+
104
+ def _wrap_unowned_calls(steps: list[Step], root_name: str) -> list[Step]:
105
+ """Model and tool calls with no flow step around them get a synthetic agent."""
106
+ unowned = [s for s in steps if s.kind in CALL_KINDS and not s.hidden and s.scope_id is None]
107
+ if not unowned:
108
+ return steps
109
+ first = min(unowned, key=_order)
110
+ running = any(s.end_ns is None for s in unowned)
111
+ root = Step(
112
+ id=f"{first.run_id}~agent",
113
+ run_id=first.run_id,
114
+ parent_id=None,
115
+ scope_id=None,
116
+ kind="agent",
117
+ type_label="agent",
118
+ name=root_name,
119
+ key="",
120
+ status="running" if running else "ok",
121
+ start_ns=first.start_ns,
122
+ end_ns=None if running else max(s.end_ns or 0 for s in unowned),
123
+ convention=first.convention,
124
+ synthetic=True,
125
+ )
126
+ for s in unowned:
127
+ s.scope_id = root.id
128
+ if s.parent_id is None:
129
+ s.parent_id = root.id
130
+ return [root, *steps]
131
+
132
+
133
+ def _order(s: Step) -> tuple[int, str]:
134
+ return (s.start_ns, s.id)
135
+
136
+
137
+ def _turn_of(call: Step, models: list[Step]) -> int:
138
+ turn = -1
139
+ for i, m in enumerate(models):
140
+ if m.end_ns is not None and m.end_ns <= call.start_ns:
141
+ turn = i
142
+ return turn
143
+
144
+
145
+ def _synthetic(agent: Step, step_id: str, name: str, inside: list[Step]) -> Step:
146
+ running = any(c.end_ns is None for c in inside)
147
+ status: StepStatus = "running" if running else "ok"
148
+ if name == "model" and not running:
149
+ status = inside[0].status
150
+ return Step(
151
+ id=step_id,
152
+ run_id=agent.run_id,
153
+ parent_id=agent.id,
154
+ scope_id=agent.id,
155
+ kind="node",
156
+ type_label=name,
157
+ name=name,
158
+ key="", # set by _rekey
159
+ status=status,
160
+ start_ns=min(c.start_ns for c in inside),
161
+ end_ns=None if running else max(c.end_ns or 0 for c in inside),
162
+ convention=agent.convention,
163
+ synthetic=True,
164
+ )
165
+
166
+
167
+ def _enclosing_call(s: Step, by_id: dict[str, Step], moved: dict[str, str]) -> str | None:
168
+ """The moved call that `s` happened inside, if any (walking visible parents)."""
169
+ parent_id = s.parent_id
170
+ while parent_id is not None and parent_id in by_id:
171
+ if parent_id in moved:
172
+ return parent_id
173
+ parent = by_id[parent_id]
174
+ if parent.kind in FLOW_KINDS:
175
+ return None # another flow step owns it
176
+ parent_id = parent.parent_id
177
+ return None
178
+
179
+
180
+ def _rekey(steps: list[Step], by_id: dict[str, Step]) -> None:
181
+ """Keys are the scope path plus the name, as the normalizer builds them; scopes
182
+ changed, so build them again."""
183
+ keys: dict[str, str] = {}
184
+
185
+ def key_of(step: Step) -> str:
186
+ if step.id not in keys:
187
+ name = step.name if step.kind in FLOW_KINDS else f"{step.kind}:{step.name}"
188
+ scope = by_id.get(step.scope_id) if step.scope_id else None
189
+ keys[step.id] = f"{key_of(scope)}/{name}" if scope else name
190
+ return keys[step.id]
191
+
192
+ for s in steps:
193
+ s.key = key_of(s)