loopview 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- loopview/__init__.py +3 -0
- loopview/app.py +252 -0
- loopview/cli.py +128 -0
- loopview/cost/__init__.py +1 -0
- loopview/cost/pricing.json +62 -0
- loopview/cost/pricing.py +74 -0
- loopview/cost/split.py +228 -0
- loopview/demo_data/flagship.otlp.jsonl +13 -0
- loopview/devtools/__init__.py +1 -0
- loopview/devtools/dump_normalized.py +211 -0
- loopview/ingest/__init__.py +1 -0
- loopview/ingest/otlp.py +305 -0
- loopview/ingest/raw.py +47 -0
- loopview/live.py +154 -0
- loopview/normalize/__init__.py +1 -0
- loopview/normalize/adapters/__init__.py +14 -0
- loopview/normalize/adapters/base.py +87 -0
- loopview/normalize/adapters/gen_ai.py +252 -0
- loopview/normalize/adapters/generic.py +22 -0
- loopview/normalize/adapters/openinference.py +367 -0
- loopview/normalize/derived_tools.py +90 -0
- loopview/normalize/loop_nodes.py +193 -0
- loopview/normalize/normalizer.py +326 -0
- loopview/normalize/schema.py +161 -0
- loopview/normalize/transitions.py +106 -0
- loopview/py.typed +0 -0
- loopview/static/assets/elk-worker.min-DfmSo98M.js +22 -0
- loopview/static/assets/index-CEGppBgk.css +1 -0
- loopview/static/assets/index-DzC96LiQ.js +21 -0
- loopview/static/assets/inter-cyrillic-ext-wght-normal-BOeWTOD4.woff2 +0 -0
- loopview/static/assets/inter-cyrillic-wght-normal-DqGufNeO.woff2 +0 -0
- loopview/static/assets/inter-greek-ext-wght-normal-DlzME5K_.woff2 +0 -0
- loopview/static/assets/inter-greek-wght-normal-CkhJZR-_.woff2 +0 -0
- loopview/static/assets/inter-latin-ext-wght-normal-DO1Apj_S.woff2 +0 -0
- loopview/static/assets/inter-latin-wght-normal-Dx4kXJAl.woff2 +0 -0
- loopview/static/assets/inter-vietnamese-wght-normal-CBcvBZtf.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-cyrillic-wght-normal-D73BlboJ.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-greek-wght-normal-Bw9x6K1M.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-latin-ext-wght-normal-DBQx-q_a.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-latin-wght-normal-B9CIFXIH.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-vietnamese-wght-normal-Bt-aOZkq.woff2 +0 -0
- loopview/static/index.html +13 -0
- loopview/store/__init__.py +1 -0
- loopview/store/capture.py +112 -0
- loopview/store/memory.py +165 -0
- loopview/tools/__init__.py +1 -0
- loopview/tools/report.py +347 -0
- loopview-0.1.0.dist-info/METADATA +72 -0
- loopview-0.1.0.dist-info/RECORD +51 -0
- loopview-0.1.0.dist-info/WHEEL +4 -0
- loopview-0.1.0.dist-info/entry_points.txt +3 -0
loopview/live.py
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
"""The live hub: keeps normalized runs up to date and pushes changes to browsers.
|
|
2
|
+
|
|
3
|
+
Flow:
|
|
4
|
+
receiver -> store.add_spans -> hub.mark(trace ids)
|
|
5
|
+
every 100 ms: for each marked run, normalize it, work out which steps changed
|
|
6
|
+
since the last push, and send one event to every connected browser.
|
|
7
|
+
|
|
8
|
+
Batching every 100 ms bounds the work when an exporter sends many small requests,
|
|
9
|
+
and makes one event per run per tick. Events are idempotent upserts, so a browser
|
|
10
|
+
can fetch a full snapshot and subscribe in any order without losing anything.
|
|
11
|
+
|
|
12
|
+
Transport is Server-Sent Events (see DECISIONS.md D7): one HTTP response that
|
|
13
|
+
stays open, each event a `data: {json}` line.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
import asyncio
|
|
17
|
+
import contextlib
|
|
18
|
+
import time
|
|
19
|
+
from collections.abc import AsyncIterator, Iterable
|
|
20
|
+
|
|
21
|
+
from loopview.cost.pricing import Pricing
|
|
22
|
+
from loopview.normalize.normalizer import normalize_run
|
|
23
|
+
from loopview.normalize.schema import NormalizedRun, RunInfo
|
|
24
|
+
from loopview.store.memory import TraceStore
|
|
25
|
+
|
|
26
|
+
FLUSH_INTERVAL_S = 0.1
|
|
27
|
+
# Runs still marked running are re-checked this often, so a run whose process
|
|
28
|
+
# died flips to finished without any new data arriving.
|
|
29
|
+
STALE_CHECK_INTERVAL_S = 2.0
|
|
30
|
+
HEARTBEAT_INTERVAL_S = 15.0
|
|
31
|
+
MAX_QUEUED_EVENTS = 1000
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class LiveHub:
|
|
35
|
+
def __init__(self, store: TraceStore, pricing: Pricing | None = None) -> None:
|
|
36
|
+
self.store = store
|
|
37
|
+
self.pricing = pricing # None: the bundled prices
|
|
38
|
+
self._dirty: set[str] = set()
|
|
39
|
+
self._cache: dict[str, NormalizedRun] = {}
|
|
40
|
+
# Per run, the JSON of each step as last pushed, to send only what changed.
|
|
41
|
+
self._pushed: dict[str, dict[str, str]] = {}
|
|
42
|
+
# Each browser's queue of SSE messages; None tells the stream to end.
|
|
43
|
+
self._subscribers: set[asyncio.Queue[str | None]] = set()
|
|
44
|
+
self._closing = False
|
|
45
|
+
|
|
46
|
+
# --- called by the receiver and the API --------------------------------------------
|
|
47
|
+
|
|
48
|
+
def mark(self, trace_ids: Iterable[str]) -> None:
|
|
49
|
+
self._dirty.update(trace_ids)
|
|
50
|
+
|
|
51
|
+
def get(self, trace_id: str) -> NormalizedRun | None:
|
|
52
|
+
if trace_id in self._dirty or trace_id not in self._cache:
|
|
53
|
+
run = self.store.get_run(trace_id)
|
|
54
|
+
if run is None:
|
|
55
|
+
return None
|
|
56
|
+
self._cache[trace_id] = normalize_run(run, pricing=self.pricing)
|
|
57
|
+
return self._cache[trace_id]
|
|
58
|
+
|
|
59
|
+
def run_infos(self) -> list[RunInfo]:
|
|
60
|
+
infos = []
|
|
61
|
+
for run in self.store.runs():
|
|
62
|
+
normalized = self.get(run.trace_id)
|
|
63
|
+
if normalized is not None:
|
|
64
|
+
infos.append(normalized.run)
|
|
65
|
+
return infos
|
|
66
|
+
|
|
67
|
+
# --- subscribers ------------------------------------------------------------------
|
|
68
|
+
|
|
69
|
+
async def events(self) -> AsyncIterator[str]:
|
|
70
|
+
"""Yield SSE-formatted events for one browser until it disconnects or the
|
|
71
|
+
server shuts down."""
|
|
72
|
+
queue: asyncio.Queue[str | None] = asyncio.Queue(MAX_QUEUED_EVENTS)
|
|
73
|
+
self._subscribers.add(queue)
|
|
74
|
+
try:
|
|
75
|
+
yield ": connected\n\n"
|
|
76
|
+
while not self._closing:
|
|
77
|
+
try:
|
|
78
|
+
message = await asyncio.wait_for(queue.get(), HEARTBEAT_INTERVAL_S)
|
|
79
|
+
except TimeoutError:
|
|
80
|
+
yield ": heartbeat\n\n" # keeps proxies from closing the connection
|
|
81
|
+
continue
|
|
82
|
+
if message is None:
|
|
83
|
+
return
|
|
84
|
+
yield message
|
|
85
|
+
finally:
|
|
86
|
+
self._subscribers.discard(queue)
|
|
87
|
+
|
|
88
|
+
def close(self) -> None:
|
|
89
|
+
"""End every open stream. Called when the server starts shutting down:
|
|
90
|
+
a stream never ends on its own, and the server waits for open
|
|
91
|
+
connections before it stops, so without this Ctrl+C would hang."""
|
|
92
|
+
self._closing = True
|
|
93
|
+
for queue in list(self._subscribers):
|
|
94
|
+
try:
|
|
95
|
+
queue.put_nowait(None)
|
|
96
|
+
except asyncio.QueueFull:
|
|
97
|
+
self._subscribers.discard(queue)
|
|
98
|
+
|
|
99
|
+
def _publish(self, payload: str) -> None:
|
|
100
|
+
for queue in list(self._subscribers):
|
|
101
|
+
try:
|
|
102
|
+
queue.put_nowait(f"data: {payload}\n\n")
|
|
103
|
+
except asyncio.QueueFull:
|
|
104
|
+
# A browser that stopped reading: drop it; it reconnects and refetches.
|
|
105
|
+
self._subscribers.discard(queue)
|
|
106
|
+
|
|
107
|
+
# --- the background loop ------------------------------------------------------------
|
|
108
|
+
|
|
109
|
+
async def run_forever(self) -> None:
|
|
110
|
+
last_stale_check = time.monotonic()
|
|
111
|
+
while True:
|
|
112
|
+
await asyncio.sleep(FLUSH_INTERVAL_S)
|
|
113
|
+
if time.monotonic() - last_stale_check > STALE_CHECK_INTERVAL_S:
|
|
114
|
+
last_stale_check = time.monotonic()
|
|
115
|
+
self.mark(t for t, n in self._cache.items() if n.run.status == "running")
|
|
116
|
+
self.flush()
|
|
117
|
+
|
|
118
|
+
def flush(self) -> None:
|
|
119
|
+
dirty, self._dirty = self._dirty, set()
|
|
120
|
+
for trace_id in dirty:
|
|
121
|
+
run = self.store.get_run(trace_id)
|
|
122
|
+
if run is None: # evicted from the ring buffer
|
|
123
|
+
self._cache.pop(trace_id, None)
|
|
124
|
+
self._pushed.pop(trace_id, None)
|
|
125
|
+
continue
|
|
126
|
+
normalized = normalize_run(run, pricing=self.pricing)
|
|
127
|
+
self._cache[trace_id] = normalized
|
|
128
|
+
pushed = self._pushed.setdefault(trace_id, {})
|
|
129
|
+
changed = []
|
|
130
|
+
for step in normalized.steps:
|
|
131
|
+
as_json = step.model_dump_json()
|
|
132
|
+
if pushed.get(step.id) != as_json:
|
|
133
|
+
pushed[step.id] = as_json
|
|
134
|
+
changed.append(as_json)
|
|
135
|
+
transitions = ",".join(t.model_dump_json() for t in normalized.transitions)
|
|
136
|
+
self._publish(
|
|
137
|
+
f'{{"type":"run.update","run":{normalized.run.model_dump_json()},'
|
|
138
|
+
f'"steps":[{",".join(changed)}],"transitions":[{transitions}]}}'
|
|
139
|
+
)
|
|
140
|
+
# Forget caches of runs the store evicted without them being marked.
|
|
141
|
+
for trace_id in [t for t in self._cache if self.store.get_run(t) is None]:
|
|
142
|
+
self._cache.pop(trace_id, None)
|
|
143
|
+
self._pushed.pop(trace_id, None)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
@contextlib.asynccontextmanager
|
|
147
|
+
async def running(hub: LiveHub) -> AsyncIterator[None]:
|
|
148
|
+
task = asyncio.create_task(hub.run_forever())
|
|
149
|
+
try:
|
|
150
|
+
yield
|
|
151
|
+
finally:
|
|
152
|
+
task.cancel()
|
|
153
|
+
with contextlib.suppress(asyncio.CancelledError):
|
|
154
|
+
await task
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Normalization: raw spans from any convention in, one small schema out."""
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""The adapter registry. The first adapter whose `matches` is true wins, so the
|
|
2
|
+
generic fallback must stay last."""
|
|
3
|
+
|
|
4
|
+
from loopview.ingest.raw import RawSpan
|
|
5
|
+
from loopview.normalize.adapters.base import Adapter
|
|
6
|
+
from loopview.normalize.adapters.gen_ai import GenAiAdapter
|
|
7
|
+
from loopview.normalize.adapters.generic import GenericAdapter
|
|
8
|
+
from loopview.normalize.adapters.openinference import OpenInferenceAdapter
|
|
9
|
+
|
|
10
|
+
ADAPTERS: list[Adapter] = [OpenInferenceAdapter(), GenAiAdapter(), GenericAdapter()]
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def adapter_for(span: RawSpan) -> Adapter:
|
|
14
|
+
return next(a for a in ADAPTERS if a.matches(span))
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""The adapter contract.
|
|
2
|
+
|
|
3
|
+
An adapter knows one span convention. For one span, it answers:
|
|
4
|
+
- does this span follow my convention? (`matches`)
|
|
5
|
+
- what is it? (`classify`, returning a Classification)
|
|
6
|
+
- if this span arrived but its parent hasn't yet, what can I tell about the
|
|
7
|
+
parent? (`infer_parent`, used to show still-running steps with a real name)
|
|
8
|
+
|
|
9
|
+
Adding a convention means writing one adapter and adding it to ADAPTERS in
|
|
10
|
+
normalize/adapters/__init__.py. Nothing else changes.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
from dataclasses import dataclass, field
|
|
15
|
+
from typing import Any, Protocol
|
|
16
|
+
|
|
17
|
+
from loopview.ingest.raw import RawSpan
|
|
18
|
+
from loopview.normalize.schema import ModelCall, StepKind, ToolCall
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class Classification:
|
|
23
|
+
kind: StepKind
|
|
24
|
+
name: str
|
|
25
|
+
type_label: str
|
|
26
|
+
hidden: bool = False
|
|
27
|
+
model: ModelCall | None = None
|
|
28
|
+
tool: ToolCall | None = None
|
|
29
|
+
input: Any = None
|
|
30
|
+
output: Any = None
|
|
31
|
+
# When True, a later pass hides this step if its parent has the same name
|
|
32
|
+
# (a subgraph span nested directly in the node that runs it).
|
|
33
|
+
collapse_into_parent: bool = False
|
|
34
|
+
attributes: dict[str, Any] = field(default_factory=dict)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass
|
|
38
|
+
class ParentHint:
|
|
39
|
+
name: str
|
|
40
|
+
kind: StepKind
|
|
41
|
+
type_label: str
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class Adapter(Protocol):
|
|
45
|
+
name: str
|
|
46
|
+
|
|
47
|
+
def matches(self, span: RawSpan) -> bool: ...
|
|
48
|
+
|
|
49
|
+
def classify(self, span: RawSpan) -> Classification: ...
|
|
50
|
+
|
|
51
|
+
def infer_parent(self, span: RawSpan) -> ParentHint | None: ...
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
# --- helpers shared by adapters ---------------------------------------------------
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def maybe_json(value: Any) -> Any:
|
|
58
|
+
"""Parse a JSON string; return anything else (or unparseable text) unchanged.
|
|
59
|
+
|
|
60
|
+
Conventions allow structured values to be recorded as JSON strings when the
|
|
61
|
+
exporter can't carry structure, so the same attribute can arrive either way."""
|
|
62
|
+
if isinstance(value, str) and value[:1] in ("{", "[", '"'):
|
|
63
|
+
try:
|
|
64
|
+
return json.loads(value)
|
|
65
|
+
except ValueError:
|
|
66
|
+
return value
|
|
67
|
+
return value
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def as_int(value: Any) -> int | None:
|
|
71
|
+
try:
|
|
72
|
+
return int(value) if value is not None else None
|
|
73
|
+
except (TypeError, ValueError):
|
|
74
|
+
return None
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def error_message(span: RawSpan) -> str | None:
|
|
78
|
+
"""The error of a failed span: the status message, or the recorded exception."""
|
|
79
|
+
if span.status_code != "error":
|
|
80
|
+
return None
|
|
81
|
+
for event in span.events:
|
|
82
|
+
if event.name == "exception":
|
|
83
|
+
message = event.attributes.get("exception.message")
|
|
84
|
+
kind = event.attributes.get("exception.type")
|
|
85
|
+
if message or kind:
|
|
86
|
+
return ": ".join(str(x) for x in (kind, message) if x)
|
|
87
|
+
return span.status_message or span.attributes.get("error.type") or "error"
|
|
@@ -0,0 +1,252 @@
|
|
|
1
|
+
"""Adapter for the OpenTelemetry GenAI semantic conventions.
|
|
2
|
+
|
|
3
|
+
Written against open-telemetry/semantic-conventions-genai at commit bcc7f9c
|
|
4
|
+
(2026-09-29), status Development. Also reads the older shape that existing
|
|
5
|
+
instrumentations still emit: messages as span events (gen_ai.user.message,
|
|
6
|
+
gen_ai.choice, ...) and gen_ai.system instead of gen_ai.provider.name.
|
|
7
|
+
|
|
8
|
+
A span is GenAI if it has gen_ai.operation.name.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from loopview.ingest.raw import RawSpan
|
|
14
|
+
from loopview.normalize.adapters.base import (
|
|
15
|
+
Classification,
|
|
16
|
+
ParentHint,
|
|
17
|
+
as_int,
|
|
18
|
+
maybe_json,
|
|
19
|
+
)
|
|
20
|
+
from loopview.normalize.schema import Message, MessagePart, ModelCall, ToolCall, Usage
|
|
21
|
+
|
|
22
|
+
MODEL_OPERATIONS = {"chat", "text_completion", "generate_content", "embeddings"}
|
|
23
|
+
TOOL_LIKE_OPERATIONS = {
|
|
24
|
+
"execute_tool",
|
|
25
|
+
"retrieval",
|
|
26
|
+
"create_memory",
|
|
27
|
+
"search_memory",
|
|
28
|
+
"update_memory",
|
|
29
|
+
"upsert_memory",
|
|
30
|
+
"delete_memory",
|
|
31
|
+
"create_memory_store",
|
|
32
|
+
"delete_memory_store",
|
|
33
|
+
}
|
|
34
|
+
# The event that carries messages when content is recorded on events.
|
|
35
|
+
DETAILS_EVENT = "gen_ai.client.inference.operation.details"
|
|
36
|
+
# Older, deprecated per-message events.
|
|
37
|
+
LEGACY_MESSAGE_EVENTS = {
|
|
38
|
+
"gen_ai.system.message": "system",
|
|
39
|
+
"gen_ai.user.message": "user",
|
|
40
|
+
"gen_ai.assistant.message": "assistant",
|
|
41
|
+
"gen_ai.tool.message": "tool",
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class GenAiAdapter:
|
|
46
|
+
name = "gen_ai"
|
|
47
|
+
|
|
48
|
+
def matches(self, span: RawSpan) -> bool:
|
|
49
|
+
return "gen_ai.operation.name" in span.attributes
|
|
50
|
+
|
|
51
|
+
def classify(self, span: RawSpan) -> Classification:
|
|
52
|
+
attrs = span.attributes
|
|
53
|
+
operation = str(attrs["gen_ai.operation.name"])
|
|
54
|
+
|
|
55
|
+
if operation == "invoke_agent":
|
|
56
|
+
name = attrs.get("gen_ai.agent.name") or _name_after_operation(span.name, operation)
|
|
57
|
+
return Classification("agent", str(name), "agent", output=_final_output(attrs))
|
|
58
|
+
if operation == "invoke_workflow":
|
|
59
|
+
name = attrs.get("gen_ai.workflow.name") or _name_after_operation(span.name, operation)
|
|
60
|
+
return Classification("agent", str(name), "workflow")
|
|
61
|
+
if operation in MODEL_OPERATIONS:
|
|
62
|
+
model = attrs.get("gen_ai.request.model") or attrs.get("gen_ai.response.model")
|
|
63
|
+
return Classification(
|
|
64
|
+
"model_call", str(model or span.name), "llm", model=_model_call(span)
|
|
65
|
+
)
|
|
66
|
+
if operation in TOOL_LIKE_OPERATIONS:
|
|
67
|
+
tool_name = attrs.get("gen_ai.tool.name") or _name_after_operation(span.name, operation)
|
|
68
|
+
label = "tool" if operation == "execute_tool" else operation
|
|
69
|
+
return Classification(
|
|
70
|
+
"tool_call",
|
|
71
|
+
str(tool_name),
|
|
72
|
+
label,
|
|
73
|
+
tool=ToolCall(
|
|
74
|
+
name=str(tool_name),
|
|
75
|
+
call_id=attrs.get("gen_ai.tool.call.id"),
|
|
76
|
+
arguments=maybe_json(attrs.get("gen_ai.tool.call.arguments")),
|
|
77
|
+
result=maybe_json(attrs.get("gen_ai.tool.call.result")),
|
|
78
|
+
),
|
|
79
|
+
)
|
|
80
|
+
# plan, create_agent, or an operation this version doesn't know yet.
|
|
81
|
+
name = attrs.get("gen_ai.agent.name") or span.name
|
|
82
|
+
return Classification("node", str(name), operation)
|
|
83
|
+
|
|
84
|
+
def infer_parent(self, span: RawSpan) -> ParentHint | None:
|
|
85
|
+
# Model and tool calls made inside an agent carry the agent's name.
|
|
86
|
+
operation = span.attributes.get("gen_ai.operation.name")
|
|
87
|
+
agent = span.attributes.get("gen_ai.agent.name")
|
|
88
|
+
if agent and operation != "invoke_agent":
|
|
89
|
+
return ParentHint(str(agent), "agent", "agent")
|
|
90
|
+
return None
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _name_after_operation(span_name: str, operation: str) -> str:
|
|
94
|
+
"""Span names look like '{operation} {name}'; take the name part."""
|
|
95
|
+
prefix = operation + " "
|
|
96
|
+
return span_name[len(prefix) :] if span_name.startswith(prefix) else span_name
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _final_output(attrs: dict[str, Any]) -> Any:
|
|
100
|
+
messages = _messages(attrs.get("gen_ai.output.messages"))
|
|
101
|
+
return _text_of(messages) if messages else None
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _model_call(span: RawSpan) -> ModelCall:
|
|
105
|
+
attrs = dict(span.attributes)
|
|
106
|
+
# Content can live on the inference details event instead of the span.
|
|
107
|
+
for event in span.events:
|
|
108
|
+
if event.name == DETAILS_EVENT:
|
|
109
|
+
for key in (
|
|
110
|
+
"gen_ai.input.messages",
|
|
111
|
+
"gen_ai.output.messages",
|
|
112
|
+
"gen_ai.system_instructions",
|
|
113
|
+
):
|
|
114
|
+
attrs.setdefault(key, event.attributes.get(key))
|
|
115
|
+
|
|
116
|
+
inputs = _messages(attrs.get("gen_ai.input.messages"))
|
|
117
|
+
outputs = _messages(attrs.get("gen_ai.output.messages"))
|
|
118
|
+
system = _parts(maybe_json(attrs.get("gen_ai.system_instructions")))
|
|
119
|
+
if system:
|
|
120
|
+
inputs = [Message(role="system", parts=system), *inputs]
|
|
121
|
+
if not inputs and not outputs:
|
|
122
|
+
inputs, outputs = _legacy_event_messages(span)
|
|
123
|
+
|
|
124
|
+
return ModelCall(
|
|
125
|
+
provider=attrs.get("gen_ai.provider.name") or attrs.get("gen_ai.system"),
|
|
126
|
+
model=attrs.get("gen_ai.response.model") or attrs.get("gen_ai.request.model"),
|
|
127
|
+
input=inputs,
|
|
128
|
+
output=outputs,
|
|
129
|
+
tool_definitions=_list(attrs.get("gen_ai.tool.definitions")),
|
|
130
|
+
usage=usage_from_attributes(attrs),
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
# Token counts. The spec's names first, then the names instrumentations are known
|
|
135
|
+
# to use instead (Pydantic AI writes gen_ai.usage.details.*, from the provider's
|
|
136
|
+
# own field names). The spec defines input_tokens as including cache reads and
|
|
137
|
+
# writes, and output_tokens as including reasoning.
|
|
138
|
+
_CACHE_READ_KEYS = (
|
|
139
|
+
"gen_ai.usage.cache_read.input_tokens",
|
|
140
|
+
"gen_ai.usage.details.cache_read_input_tokens",
|
|
141
|
+
)
|
|
142
|
+
_CACHE_WRITE_KEYS = (
|
|
143
|
+
"gen_ai.usage.cache_write.input_tokens",
|
|
144
|
+
"gen_ai.usage.details.cache_creation_input_tokens", # Pydantic AI
|
|
145
|
+
"gen_ai.usage.cache_creation.input_tokens", # OpenLLMetry
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def usage_from_attributes(attrs: dict[str, Any]) -> Usage | None:
|
|
150
|
+
usage = Usage(
|
|
151
|
+
input_tokens=as_int(attrs.get("gen_ai.usage.input_tokens")),
|
|
152
|
+
output_tokens=as_int(attrs.get("gen_ai.usage.output_tokens")),
|
|
153
|
+
cache_read_tokens=_first_int(attrs, _CACHE_READ_KEYS),
|
|
154
|
+
cache_write_tokens=_first_int(attrs, _CACHE_WRITE_KEYS),
|
|
155
|
+
reasoning_tokens=as_int(attrs.get("gen_ai.usage.reasoning.output_tokens")),
|
|
156
|
+
)
|
|
157
|
+
return usage if usage.model_dump(exclude_none=True) else None
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _first_int(attrs: dict[str, Any], keys: tuple[str, ...]) -> int | None:
|
|
161
|
+
for key in keys:
|
|
162
|
+
value = as_int(attrs.get(key))
|
|
163
|
+
if value is not None:
|
|
164
|
+
return value
|
|
165
|
+
return None
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _list(value: Any) -> list[Any]:
|
|
169
|
+
value = maybe_json(value)
|
|
170
|
+
return value if isinstance(value, list) else []
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _messages(value: Any) -> list[Message]:
|
|
174
|
+
value = maybe_json(value)
|
|
175
|
+
if not isinstance(value, list):
|
|
176
|
+
return []
|
|
177
|
+
return [
|
|
178
|
+
Message(role=str(m.get("role", "unknown")), parts=_parts(m.get("parts")))
|
|
179
|
+
for m in value
|
|
180
|
+
if isinstance(m, dict)
|
|
181
|
+
]
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _parts(value: Any) -> list[MessagePart]:
|
|
185
|
+
if isinstance(value, str):
|
|
186
|
+
return [MessagePart(type="text", text=value)]
|
|
187
|
+
if not isinstance(value, list):
|
|
188
|
+
return []
|
|
189
|
+
parts = []
|
|
190
|
+
for p in value:
|
|
191
|
+
if not isinstance(p, dict):
|
|
192
|
+
continue
|
|
193
|
+
kind = p.get("type")
|
|
194
|
+
if kind == "text":
|
|
195
|
+
parts.append(MessagePart(type="text", text=str(p.get("content", ""))))
|
|
196
|
+
elif kind == "tool_call":
|
|
197
|
+
parts.append(
|
|
198
|
+
MessagePart(
|
|
199
|
+
type="tool_call",
|
|
200
|
+
id=p.get("id"),
|
|
201
|
+
name=p.get("name"),
|
|
202
|
+
arguments=maybe_json(p.get("arguments")),
|
|
203
|
+
)
|
|
204
|
+
)
|
|
205
|
+
elif kind == "reasoning":
|
|
206
|
+
parts.append(MessagePart(type="reasoning", text=str(p.get("content", ""))))
|
|
207
|
+
elif kind == "tool_call_response":
|
|
208
|
+
parts.append(
|
|
209
|
+
MessagePart(
|
|
210
|
+
type="tool_result",
|
|
211
|
+
id=p.get("id"),
|
|
212
|
+
result=maybe_json(p.get("response", p.get("result"))),
|
|
213
|
+
# Not in the spec; read when an instrumentation records it.
|
|
214
|
+
is_error=p.get("is_error") if isinstance(p.get("is_error"), bool) else None,
|
|
215
|
+
)
|
|
216
|
+
)
|
|
217
|
+
else:
|
|
218
|
+
# reasoning, blob, uri, file and future part types: keep them readable.
|
|
219
|
+
content = p.get("content")
|
|
220
|
+
parts.append(
|
|
221
|
+
MessagePart(type="other", name=kind, text=None if content is None else str(content))
|
|
222
|
+
)
|
|
223
|
+
return parts
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _legacy_event_messages(span: RawSpan) -> tuple[list[Message], list[Message]]:
|
|
227
|
+
inputs, outputs = [], []
|
|
228
|
+
for event in span.events:
|
|
229
|
+
a = event.attributes
|
|
230
|
+
if event.name in LEGACY_MESSAGE_EVENTS:
|
|
231
|
+
content = maybe_json(a.get("content"))
|
|
232
|
+
inputs.append(
|
|
233
|
+
Message(
|
|
234
|
+
role=LEGACY_MESSAGE_EVENTS[event.name],
|
|
235
|
+
parts=[MessagePart(type="text", text=_as_text(content))],
|
|
236
|
+
)
|
|
237
|
+
)
|
|
238
|
+
elif event.name == "gen_ai.choice":
|
|
239
|
+
message = maybe_json(a.get("message"))
|
|
240
|
+
content = message.get("content") if isinstance(message, dict) else message
|
|
241
|
+
outputs.append(
|
|
242
|
+
Message(role="assistant", parts=[MessagePart(type="text", text=_as_text(content))])
|
|
243
|
+
)
|
|
244
|
+
return inputs, outputs
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def _as_text(value: Any) -> str:
|
|
248
|
+
return value if isinstance(value, str) else ("" if value is None else str(value))
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _text_of(messages: list[Message]) -> str:
|
|
252
|
+
return "\n".join(p.text for m in messages for p in m.parts if p.type == "text" and p.text)
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""Fallback adapter: any span no other adapter recognised.
|
|
2
|
+
|
|
3
|
+
It becomes an `unknown` step that keeps its raw attributes, so nothing is ever
|
|
4
|
+
dropped. Whether it is drawn is decided in the normalizer (unknown spans inside a
|
|
5
|
+
model or tool call, such as HTTP client spans, are implementation detail).
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from loopview.ingest.raw import RawSpan
|
|
9
|
+
from loopview.normalize.adapters.base import Classification, ParentHint
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class GenericAdapter:
|
|
13
|
+
name = "generic"
|
|
14
|
+
|
|
15
|
+
def matches(self, span: RawSpan) -> bool:
|
|
16
|
+
return True
|
|
17
|
+
|
|
18
|
+
def classify(self, span: RawSpan) -> Classification:
|
|
19
|
+
return Classification("unknown", span.name, span.kind, attributes=dict(span.attributes))
|
|
20
|
+
|
|
21
|
+
def infer_parent(self, span: RawSpan) -> ParentHint | None:
|
|
22
|
+
return None
|