loopview 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- loopview/__init__.py +3 -0
- loopview/app.py +252 -0
- loopview/cli.py +128 -0
- loopview/cost/__init__.py +1 -0
- loopview/cost/pricing.json +62 -0
- loopview/cost/pricing.py +74 -0
- loopview/cost/split.py +228 -0
- loopview/demo_data/flagship.otlp.jsonl +13 -0
- loopview/devtools/__init__.py +1 -0
- loopview/devtools/dump_normalized.py +211 -0
- loopview/ingest/__init__.py +1 -0
- loopview/ingest/otlp.py +305 -0
- loopview/ingest/raw.py +47 -0
- loopview/live.py +154 -0
- loopview/normalize/__init__.py +1 -0
- loopview/normalize/adapters/__init__.py +14 -0
- loopview/normalize/adapters/base.py +87 -0
- loopview/normalize/adapters/gen_ai.py +252 -0
- loopview/normalize/adapters/generic.py +22 -0
- loopview/normalize/adapters/openinference.py +367 -0
- loopview/normalize/derived_tools.py +90 -0
- loopview/normalize/loop_nodes.py +193 -0
- loopview/normalize/normalizer.py +326 -0
- loopview/normalize/schema.py +161 -0
- loopview/normalize/transitions.py +106 -0
- loopview/py.typed +0 -0
- loopview/static/assets/elk-worker.min-DfmSo98M.js +22 -0
- loopview/static/assets/index-CEGppBgk.css +1 -0
- loopview/static/assets/index-DzC96LiQ.js +21 -0
- loopview/static/assets/inter-cyrillic-ext-wght-normal-BOeWTOD4.woff2 +0 -0
- loopview/static/assets/inter-cyrillic-wght-normal-DqGufNeO.woff2 +0 -0
- loopview/static/assets/inter-greek-ext-wght-normal-DlzME5K_.woff2 +0 -0
- loopview/static/assets/inter-greek-wght-normal-CkhJZR-_.woff2 +0 -0
- loopview/static/assets/inter-latin-ext-wght-normal-DO1Apj_S.woff2 +0 -0
- loopview/static/assets/inter-latin-wght-normal-Dx4kXJAl.woff2 +0 -0
- loopview/static/assets/inter-vietnamese-wght-normal-CBcvBZtf.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-cyrillic-wght-normal-D73BlboJ.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-greek-wght-normal-Bw9x6K1M.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-latin-ext-wght-normal-DBQx-q_a.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-latin-wght-normal-B9CIFXIH.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-vietnamese-wght-normal-Bt-aOZkq.woff2 +0 -0
- loopview/static/index.html +13 -0
- loopview/store/__init__.py +1 -0
- loopview/store/capture.py +112 -0
- loopview/store/memory.py +165 -0
- loopview/tools/__init__.py +1 -0
- loopview/tools/report.py +347 -0
- loopview-0.1.0.dist-info/METADATA +72 -0
- loopview-0.1.0.dist-info/RECORD +51 -0
- loopview-0.1.0.dist-info/WHEEL +4 -0
- loopview-0.1.0.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Developer tools, not used at runtime."""
|
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
"""Write the normalized form of every fixture as JSON.
|
|
2
|
+
|
|
3
|
+
Used for the UI tests, and for the hosted demo (a static build of the UI that
|
|
4
|
+
reads recorded runs from files instead of a server). Also writes tools.json, the
|
|
5
|
+
Tools tab's report.
|
|
6
|
+
|
|
7
|
+
Most fixtures hold one run and become `<name>.json`. A fixture with several runs
|
|
8
|
+
(the MCP tools study) becomes `<name>-01.json`, `<name>-02.json`... but only for
|
|
9
|
+
the demo (--index): the UI tests don't use them, and they are several MB.
|
|
10
|
+
|
|
11
|
+
With --index (the demo), only the fixtures in DEMO_RUNS and DEMO_TOOLS_ONLY are
|
|
12
|
+
written. index.json lists each run file with its RunInfo, so the demo can show
|
|
13
|
+
the run list without downloading every run. DEMO_RUNS are the curated examples
|
|
14
|
+
shown in the run list, each with a short description of the agent and its task;
|
|
15
|
+
DEMO_TOOLS_ONLY runs are not listed but feed the Tools tab, whose links can still
|
|
16
|
+
open them. Without --index, tools.json covers every fixture.
|
|
17
|
+
|
|
18
|
+
Usage (from server/):
|
|
19
|
+
uv run python -m loopview.devtools.dump_normalized ../ui/src/test-data
|
|
20
|
+
uv run python -m loopview.devtools.dump_normalized ../ui/pages-public/demo --index
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
import json
|
|
24
|
+
import sys
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
from typing import Any
|
|
27
|
+
|
|
28
|
+
from loopview.normalize.normalizer import normalize_run
|
|
29
|
+
from loopview.normalize.schema import NormalizedRun
|
|
30
|
+
from loopview.store.capture import load_capture
|
|
31
|
+
from loopview.store.memory import TraceStore
|
|
32
|
+
from loopview.tools.report import tools_report
|
|
33
|
+
|
|
34
|
+
REPO = Path(__file__).resolve().parents[4]
|
|
35
|
+
FIXTURES = REPO / "fixtures"
|
|
36
|
+
MCP_STUDY_TASKS = REPO / "examples" / "mcp_tools_study_tasks.txt"
|
|
37
|
+
|
|
38
|
+
# The examples the hosted demo lists, in order, with what the demo says about each.
|
|
39
|
+
# `prompt` is the task the agent was given, as written in the example's source.
|
|
40
|
+
DEMO_RUNS: dict[str, dict[str, Any]] = {
|
|
41
|
+
"flagship": {
|
|
42
|
+
"title": "Database comparison team",
|
|
43
|
+
"framework": "LangGraph",
|
|
44
|
+
"source": "examples/demo/flagship.py",
|
|
45
|
+
"summary": (
|
|
46
|
+
"A supervisor splits the question into three briefs. Three analyst agents research "
|
|
47
|
+
"benchmarks, operations and ecosystem in parallel, each with its own tools. Their "
|
|
48
|
+
"notes are merged, a critic sends the draft back once, and a writer produces the "
|
|
49
|
+
"answer."
|
|
50
|
+
),
|
|
51
|
+
"prompt": (
|
|
52
|
+
"Compare PostgreSQL, SQLite and DuckDB for a small internal analytics dashboard: "
|
|
53
|
+
"5 users, about 20 GB of event data, loaded once a day. Recommend one."
|
|
54
|
+
),
|
|
55
|
+
"look_for": [
|
|
56
|
+
"the fan-out to three analysts running at the same time",
|
|
57
|
+
"fetch_repo_stats failing on its first call, then the retry",
|
|
58
|
+
"the critic's loop back to synthesize, then the handoff to the writer",
|
|
59
|
+
"the Cost tab: a 4,300-token handbook read from the prompt cache",
|
|
60
|
+
],
|
|
61
|
+
},
|
|
62
|
+
"multi_agent_pydantic": {
|
|
63
|
+
"title": "Laptop advice, multi-agent",
|
|
64
|
+
"framework": "Pydantic AI",
|
|
65
|
+
"source": "examples/multi_agent_pydantic.py",
|
|
66
|
+
"summary": (
|
|
67
|
+
"A coordinator agent delegates to two researchers (specs and reviews) that run in "
|
|
68
|
+
"parallel behind one tool call, then hands its notes to a writer agent."
|
|
69
|
+
),
|
|
70
|
+
"prompt": (
|
|
71
|
+
"A student needs a laptop for programming and light video editing, budget 1300 EUR. "
|
|
72
|
+
"Compare the Aster Pro 14 and the Kestrel Air 15 and recommend one."
|
|
73
|
+
),
|
|
74
|
+
"look_for": [
|
|
75
|
+
"agents nested inside a tool call (delegation)",
|
|
76
|
+
"the two researchers overlapping in time",
|
|
77
|
+
"the handoff from the coordinator to the writer",
|
|
78
|
+
],
|
|
79
|
+
},
|
|
80
|
+
"langgraph_router": {
|
|
81
|
+
"title": "Date question router",
|
|
82
|
+
"framework": "LangGraph",
|
|
83
|
+
"source": "examples/langgraph_router.py",
|
|
84
|
+
"summary": (
|
|
85
|
+
"A classifier routes the question to a tool-using agent. The agent loops with its date "
|
|
86
|
+
"tools, then a strict reviewer asks for one revision and sends it back."
|
|
87
|
+
),
|
|
88
|
+
"prompt": (
|
|
89
|
+
"How many days are there from 2026-09-30 until the next February 29th, "
|
|
90
|
+
"and what day of the week will that February 29th be?"
|
|
91
|
+
),
|
|
92
|
+
"look_for": [
|
|
93
|
+
"the conditional route out of classify",
|
|
94
|
+
"the agent and tools loop",
|
|
95
|
+
"the reviewer's loop back to an earlier node",
|
|
96
|
+
],
|
|
97
|
+
},
|
|
98
|
+
"failing_tools_pydantic": {
|
|
99
|
+
"title": "Shop assistant, failing tools",
|
|
100
|
+
"framework": "Pydantic AI + MCP",
|
|
101
|
+
"source": "examples/failing_tools_pydantic.py",
|
|
102
|
+
"summary": (
|
|
103
|
+
"A support agent with an order tool and an inventory MCP server. The question is "
|
|
104
|
+
"worded so both tools fail first: the agent has to read the errors and recover."
|
|
105
|
+
),
|
|
106
|
+
"prompt": (
|
|
107
|
+
"What is the status of order 1234, and how many units of SKU-0099 (the blue mug) "
|
|
108
|
+
"are in stock?"
|
|
109
|
+
),
|
|
110
|
+
"look_for": [
|
|
111
|
+
"red tool calls: a ModelRetry and an MCP isError result",
|
|
112
|
+
"what the model does next: fixes its arguments, or switches tool",
|
|
113
|
+
"the Tools tab, where these recoveries are counted",
|
|
114
|
+
],
|
|
115
|
+
},
|
|
116
|
+
"react_anthropic": {
|
|
117
|
+
"title": "Trip budget, hand-rolled loop",
|
|
118
|
+
"framework": "Anthropic SDK",
|
|
119
|
+
"source": "examples/react_anthropic.py",
|
|
120
|
+
"summary": (
|
|
121
|
+
"A plain ReAct loop written by hand on the Anthropic SDK, traced with the "
|
|
122
|
+
"OpenTelemetry GenAI conventions. Extended thinking is on, so the model's reasoning "
|
|
123
|
+
"is recorded."
|
|
124
|
+
),
|
|
125
|
+
"prompt": (
|
|
126
|
+
"I'm planning 3 nights in Lisbon and 2 nights in Porto. Look up the average "
|
|
127
|
+
"hotel price per night in each city, then tell me the total in EUR and in USD."
|
|
128
|
+
),
|
|
129
|
+
"look_for": [
|
|
130
|
+
"the model's thinking, shown inside each model call",
|
|
131
|
+
"several tool calls asked for in one model turn",
|
|
132
|
+
],
|
|
133
|
+
},
|
|
134
|
+
}
|
|
135
|
+
# Not listed (20 short tasks against GitHub's MCP server), but the Tools tab is built on them.
|
|
136
|
+
DEMO_TOOLS_ONLY = ["mcp_tools_study"]
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def tools_only_about(name: str, run: NormalizedRun) -> dict[str, Any] | None:
|
|
140
|
+
"""What the demo says about an unlisted run that a Tools tab link opened."""
|
|
141
|
+
if name != "mcp_tools_study" or not run.run.name.startswith("task "):
|
|
142
|
+
return None # the study's first run only opens the MCP connection
|
|
143
|
+
lines = MCP_STUDY_TASKS.read_text(encoding="utf-8").splitlines()
|
|
144
|
+
tasks = [line for line in lines if line.strip() and not line.startswith("#")]
|
|
145
|
+
number = int(run.run.name.removeprefix("task "))
|
|
146
|
+
return {
|
|
147
|
+
"title": f"GitHub MCP study, task {number}",
|
|
148
|
+
"framework": "Pydantic AI + GitHub MCP",
|
|
149
|
+
"source": "examples/mcp_tools_study.py",
|
|
150
|
+
"summary": (
|
|
151
|
+
"One of 20 read-only questions a Pydantic AI agent answered with the tools of GitHub's "
|
|
152
|
+
"official MCP server. Together, these runs feed the Tools tab."
|
|
153
|
+
),
|
|
154
|
+
"prompt": tasks[number - 1],
|
|
155
|
+
"look_for": [
|
|
156
|
+
"which of the server's many tools the model picks",
|
|
157
|
+
"how big the tool results are, in the Cost tab",
|
|
158
|
+
],
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def main() -> None:
|
|
163
|
+
out = Path(sys.argv[1])
|
|
164
|
+
demo = "--index" in sys.argv
|
|
165
|
+
out.mkdir(parents=True, exist_ok=True)
|
|
166
|
+
all_runs: list[NormalizedRun] = []
|
|
167
|
+
index: list[tuple[int, str, NormalizedRun, str]] = [] # (demo position, file, run, fixture)
|
|
168
|
+
for fixture in sorted(FIXTURES.glob("*.otlp.jsonl")):
|
|
169
|
+
name = fixture.name.removesuffix(".otlp.jsonl")
|
|
170
|
+
if demo and name not in DEMO_RUNS and name not in DEMO_TOOLS_ONLY:
|
|
171
|
+
continue
|
|
172
|
+
store = TraceStore()
|
|
173
|
+
load_capture(store, fixture)
|
|
174
|
+
runs = [
|
|
175
|
+
normalize_run(run, now_ns=run.last_received_ns)
|
|
176
|
+
for run in sorted(store.runs(), key=lambda r: r.first_received_ns)
|
|
177
|
+
]
|
|
178
|
+
all_runs += runs
|
|
179
|
+
if len(runs) > 1 and not demo:
|
|
180
|
+
print(f"skipped {name} ({len(runs)} runs: written for the demo only)")
|
|
181
|
+
continue
|
|
182
|
+
position = list(DEMO_RUNS).index(name) if name in DEMO_RUNS else len(DEMO_RUNS)
|
|
183
|
+
for i, normalized in enumerate(runs, 1):
|
|
184
|
+
file = f"{name}.json" if len(runs) == 1 else f"{name}-{i:02d}.json"
|
|
185
|
+
(out / file).write_text(normalized.model_dump_json(), encoding="utf-8")
|
|
186
|
+
index.append((position, file, normalized, name))
|
|
187
|
+
print(f"wrote {name} ({len(runs)} run{'s' if len(runs) > 1 else ''})")
|
|
188
|
+
(out / "tools.json").write_text(json.dumps(tools_report(all_runs)), encoding="utf-8")
|
|
189
|
+
print("wrote tools.json")
|
|
190
|
+
if demo:
|
|
191
|
+
missing = [
|
|
192
|
+
name for name in [*DEMO_RUNS, *DEMO_TOOLS_ONLY] if not any(e[3] == name for e in index)
|
|
193
|
+
]
|
|
194
|
+
if missing:
|
|
195
|
+
sys.exit(f"no fixture for {', '.join(missing)}")
|
|
196
|
+
index.sort(key=lambda entry: entry[0]) # stable: a fixture's runs keep their order
|
|
197
|
+
entries = [
|
|
198
|
+
{
|
|
199
|
+
"file": file,
|
|
200
|
+
"run": n.run.model_dump(mode="json"),
|
|
201
|
+
"listed": name in DEMO_RUNS,
|
|
202
|
+
"about": DEMO_RUNS.get(name) or tools_only_about(name, n),
|
|
203
|
+
}
|
|
204
|
+
for _, file, n, name in index
|
|
205
|
+
]
|
|
206
|
+
(out / "index.json").write_text(json.dumps(entries), encoding="utf-8")
|
|
207
|
+
print("wrote index.json")
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
if __name__ == "__main__":
|
|
211
|
+
main()
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Ingestion: OTLP/HTTP request bodies in, RawSpans out."""
|
loopview/ingest/otlp.py
ADDED
|
@@ -0,0 +1,305 @@
|
|
|
1
|
+
"""Decode OTLP/HTTP trace export requests into RawSpans.
|
|
2
|
+
|
|
3
|
+
Follows the OTLP specification v1.11.0 (https://opentelemetry.io/docs/specs/otlp/).
|
|
4
|
+
|
|
5
|
+
Both encodings go through one path: JSON is parsed into the same protobuf message
|
|
6
|
+
as the binary encoding, then a single function converts the message to RawSpans.
|
|
7
|
+
That works because OTLP/JSON is the standard protobuf JSON mapping with one
|
|
8
|
+
exception: trace and span ids are hex strings instead of base64. We rewrite those
|
|
9
|
+
ids before parsing and let protobuf handle everything else (camelCase keys, enums
|
|
10
|
+
as integers, 64-bit integers as strings, unknown fields ignored).
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import base64
|
|
14
|
+
import json
|
|
15
|
+
import zlib
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
from google.protobuf import json_format
|
|
19
|
+
from opentelemetry.proto.collector.trace.v1.trace_service_pb2 import (
|
|
20
|
+
ExportTraceServiceRequest,
|
|
21
|
+
ExportTraceServiceResponse,
|
|
22
|
+
)
|
|
23
|
+
from opentelemetry.proto.common.v1.common_pb2 import AnyValue, KeyValue
|
|
24
|
+
from opentelemetry.proto.trace.v1.trace_pb2 import Span, Status
|
|
25
|
+
|
|
26
|
+
from loopview.ingest.raw import RawSpan, SpanEvent, SpanKind, SpanLink, StatusCode
|
|
27
|
+
|
|
28
|
+
PROTOBUF = "application/x-protobuf"
|
|
29
|
+
JSON = "application/json"
|
|
30
|
+
|
|
31
|
+
# Largest request body we accept, after decompression. Big enough for thousands of
|
|
32
|
+
# spans with full prompts; small enough that a bad request can't eat all memory.
|
|
33
|
+
MAX_BODY_BYTES = 64 * 1024 * 1024
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class OtlpDecodeError(ValueError):
|
|
37
|
+
"""The request could not be decoded. Maps to HTTP 400 (not retryable)."""
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class UnsupportedContentType(OtlpDecodeError):
|
|
41
|
+
"""Neither protobuf nor JSON. Maps to HTTP 415."""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class BodyTooLarge(OtlpDecodeError):
|
|
45
|
+
"""Maps to HTTP 413 (not retryable)."""
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def decode_request(
|
|
49
|
+
body: bytes, content_type: str | None, content_encoding: str | None = None
|
|
50
|
+
) -> list[RawSpan]:
|
|
51
|
+
"""Decode one POST /v1/traces body into RawSpans."""
|
|
52
|
+
return decode_body(decompress(body, content_encoding), content_type)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def decode_body(body: bytes, content_type: str | None) -> list[RawSpan]:
|
|
56
|
+
"""Decode an already decompressed body into RawSpans."""
|
|
57
|
+
media_type = (content_type or "").split(";")[0].strip().lower()
|
|
58
|
+
if media_type == PROTOBUF:
|
|
59
|
+
request = _parse_protobuf(body)
|
|
60
|
+
elif media_type == JSON:
|
|
61
|
+
request = _parse_json(body)
|
|
62
|
+
else:
|
|
63
|
+
raise UnsupportedContentType(f"unsupported content type: {content_type!r}")
|
|
64
|
+
return request_to_spans(request)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def encode_success_response(content_type: str | None) -> tuple[bytes, str]:
|
|
68
|
+
"""An empty ExportTraceServiceResponse in the same encoding as the request."""
|
|
69
|
+
response = ExportTraceServiceResponse()
|
|
70
|
+
if (content_type or "").lower().startswith(JSON):
|
|
71
|
+
return json_format.MessageToJson(response).encode(), JSON
|
|
72
|
+
return response.SerializeToString(), PROTOBUF
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def request_to_spans(request: ExportTraceServiceRequest) -> list[RawSpan]:
|
|
76
|
+
spans: list[RawSpan] = []
|
|
77
|
+
for resource_spans in request.resource_spans:
|
|
78
|
+
resource_attributes = _attributes(resource_spans.resource.attributes)
|
|
79
|
+
for scope_spans in resource_spans.scope_spans:
|
|
80
|
+
scope = scope_spans.scope
|
|
81
|
+
for span in scope_spans.spans:
|
|
82
|
+
spans.append(_span(span, resource_attributes, scope.name, scope.version))
|
|
83
|
+
return spans
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
# --- decoding the body ----------------------------------------------------------
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def decompress(body: bytes, content_encoding: str | None) -> bytes:
|
|
90
|
+
"""Undo Content-Encoding (identity or gzip), enforcing MAX_BODY_BYTES."""
|
|
91
|
+
encoding = (content_encoding or "identity").strip().lower()
|
|
92
|
+
if encoding == "identity":
|
|
93
|
+
result = body
|
|
94
|
+
elif encoding == "gzip":
|
|
95
|
+
# wbits=16+MAX_WBITS means "expect a gzip header". max_length stops a small
|
|
96
|
+
# compressed body from expanding into gigabytes.
|
|
97
|
+
decompressor = zlib.decompressobj(16 + zlib.MAX_WBITS)
|
|
98
|
+
try:
|
|
99
|
+
result = decompressor.decompress(body, MAX_BODY_BYTES + 1)
|
|
100
|
+
except zlib.error as exc:
|
|
101
|
+
raise OtlpDecodeError(f"invalid gzip body: {exc}") from exc
|
|
102
|
+
else:
|
|
103
|
+
raise OtlpDecodeError(f"unsupported content encoding: {content_encoding!r}")
|
|
104
|
+
if len(result) > MAX_BODY_BYTES:
|
|
105
|
+
raise BodyTooLarge(f"request body larger than {MAX_BODY_BYTES} bytes")
|
|
106
|
+
return result
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _parse_protobuf(body: bytes) -> ExportTraceServiceRequest:
|
|
110
|
+
request = ExportTraceServiceRequest()
|
|
111
|
+
try:
|
|
112
|
+
request.ParseFromString(body)
|
|
113
|
+
except Exception as exc: # protobuf raises DecodeError, but be safe
|
|
114
|
+
raise OtlpDecodeError(f"invalid protobuf body: {exc}") from exc
|
|
115
|
+
return request
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _parse_json(body: bytes) -> ExportTraceServiceRequest:
|
|
119
|
+
try:
|
|
120
|
+
data = json.loads(body)
|
|
121
|
+
_hex_ids_to_base64(data)
|
|
122
|
+
return json_format.ParseDict(data, ExportTraceServiceRequest(), ignore_unknown_fields=True)
|
|
123
|
+
except (ValueError, TypeError, json_format.ParseError) as exc:
|
|
124
|
+
raise OtlpDecodeError(f"invalid JSON body: {exc}") from exc
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
# OTLP/JSON fields that hold hex ids. Protobuf's JSON parser expects base64 for bytes.
|
|
128
|
+
_ID_FIELDS = ("traceId", "spanId", "parentSpanId")
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _hex_ids_to_base64(data: Any) -> None:
|
|
132
|
+
"""Rewrite hex trace/span ids in place, on spans and on span links."""
|
|
133
|
+
for resource_spans in data.get("resourceSpans", []):
|
|
134
|
+
for scope_spans in resource_spans.get("scopeSpans", []):
|
|
135
|
+
for span in scope_spans.get("spans", []):
|
|
136
|
+
_convert_ids(span)
|
|
137
|
+
for link in span.get("links", []):
|
|
138
|
+
_convert_ids(link)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _convert_ids(obj: dict[str, Any]) -> None:
|
|
142
|
+
for field in _ID_FIELDS:
|
|
143
|
+
value = obj.get(field)
|
|
144
|
+
if value:
|
|
145
|
+
obj[field] = base64.b64encode(bytes.fromhex(value)).decode()
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
# --- protobuf message to RawSpan ------------------------------------------------
|
|
149
|
+
|
|
150
|
+
_KINDS: dict[int, SpanKind] = {
|
|
151
|
+
Span.SPAN_KIND_UNSPECIFIED: "unspecified",
|
|
152
|
+
Span.SPAN_KIND_INTERNAL: "internal",
|
|
153
|
+
Span.SPAN_KIND_SERVER: "server",
|
|
154
|
+
Span.SPAN_KIND_CLIENT: "client",
|
|
155
|
+
Span.SPAN_KIND_PRODUCER: "producer",
|
|
156
|
+
Span.SPAN_KIND_CONSUMER: "consumer",
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
_STATUS: dict[int, StatusCode] = {
|
|
160
|
+
Status.STATUS_CODE_UNSET: "unset",
|
|
161
|
+
Status.STATUS_CODE_OK: "ok",
|
|
162
|
+
Status.STATUS_CODE_ERROR: "error",
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _span(
|
|
167
|
+
span: Span, resource_attributes: dict[str, Any], scope_name: str, scope_version: str
|
|
168
|
+
) -> RawSpan:
|
|
169
|
+
return RawSpan(
|
|
170
|
+
trace_id=span.trace_id.hex(),
|
|
171
|
+
span_id=span.span_id.hex(),
|
|
172
|
+
parent_span_id=span.parent_span_id.hex() or None,
|
|
173
|
+
name=span.name,
|
|
174
|
+
kind=_KINDS.get(span.kind, "unspecified"),
|
|
175
|
+
start_time_unix_nano=span.start_time_unix_nano,
|
|
176
|
+
end_time_unix_nano=span.end_time_unix_nano,
|
|
177
|
+
attributes=_attributes(span.attributes),
|
|
178
|
+
events=[
|
|
179
|
+
SpanEvent(
|
|
180
|
+
name=event.name,
|
|
181
|
+
time_unix_nano=event.time_unix_nano,
|
|
182
|
+
attributes=_attributes(event.attributes),
|
|
183
|
+
)
|
|
184
|
+
for event in span.events
|
|
185
|
+
],
|
|
186
|
+
links=[
|
|
187
|
+
SpanLink(
|
|
188
|
+
trace_id=link.trace_id.hex(),
|
|
189
|
+
span_id=link.span_id.hex(),
|
|
190
|
+
attributes=_attributes(link.attributes),
|
|
191
|
+
)
|
|
192
|
+
for link in span.links
|
|
193
|
+
],
|
|
194
|
+
status_code=_STATUS.get(span.status.code, "unset"),
|
|
195
|
+
status_message=span.status.message,
|
|
196
|
+
resource_attributes=resource_attributes,
|
|
197
|
+
scope_name=scope_name,
|
|
198
|
+
scope_version=scope_version,
|
|
199
|
+
)
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def _attributes(key_values: "list[KeyValue] | Any") -> dict[str, Any]:
|
|
203
|
+
return {kv.key: _value(kv.value) for kv in key_values}
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _value(value: AnyValue) -> Any:
|
|
207
|
+
"""Convert an OTLP AnyValue into a plain Python value."""
|
|
208
|
+
kind = value.WhichOneof("value")
|
|
209
|
+
if kind is None:
|
|
210
|
+
return None
|
|
211
|
+
if kind == "array_value":
|
|
212
|
+
return [_value(v) for v in value.array_value.values]
|
|
213
|
+
if kind == "kvlist_value":
|
|
214
|
+
return _attributes(value.kvlist_value.values)
|
|
215
|
+
if kind == "bytes_value":
|
|
216
|
+
# Keep RawSpans JSON-friendly: bytes become base64 text.
|
|
217
|
+
return base64.b64encode(value.bytes_value).decode()
|
|
218
|
+
return getattr(value, kind) # string_value, bool_value, int_value, double_value
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
# --- RawSpan back to OTLP/JSON (for exporting a run) ----------------------------------
|
|
222
|
+
|
|
223
|
+
_KIND_NUMBERS = {name: number for number, name in _KINDS.items()}
|
|
224
|
+
_STATUS_NUMBERS = {name: number for number, name in _STATUS.items()}
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def spans_to_otlp_json(spans: list[RawSpan]) -> dict[str, Any]:
|
|
228
|
+
"""Encode RawSpans as an OTLP/JSON ExportTraceServiceRequest.
|
|
229
|
+
|
|
230
|
+
The inverse of decoding, so an exported run can be imported again through
|
|
231
|
+
the normal receiver path. (Bytes attributes come back as base64 strings.)"""
|
|
232
|
+
groups: dict[tuple[str, str, str], list[RawSpan]] = {}
|
|
233
|
+
for span in spans:
|
|
234
|
+
key = (
|
|
235
|
+
json.dumps(span.resource_attributes, sort_keys=True),
|
|
236
|
+
span.scope_name,
|
|
237
|
+
span.scope_version,
|
|
238
|
+
)
|
|
239
|
+
groups.setdefault(key, []).append(span)
|
|
240
|
+
resource_spans = []
|
|
241
|
+
for (resource_json, scope_name, scope_version), group in groups.items():
|
|
242
|
+
resource_spans.append(
|
|
243
|
+
{
|
|
244
|
+
"resource": {"attributes": _kv_json(json.loads(resource_json))},
|
|
245
|
+
"scopeSpans": [
|
|
246
|
+
{
|
|
247
|
+
"scope": {"name": scope_name, "version": scope_version},
|
|
248
|
+
"spans": [_span_json(s) for s in group],
|
|
249
|
+
}
|
|
250
|
+
],
|
|
251
|
+
}
|
|
252
|
+
)
|
|
253
|
+
return {"resourceSpans": resource_spans}
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def _span_json(span: RawSpan) -> dict[str, Any]:
|
|
257
|
+
result: dict[str, Any] = {
|
|
258
|
+
"traceId": span.trace_id,
|
|
259
|
+
"spanId": span.span_id,
|
|
260
|
+
"name": span.name,
|
|
261
|
+
"kind": _KIND_NUMBERS[span.kind],
|
|
262
|
+
"startTimeUnixNano": str(span.start_time_unix_nano),
|
|
263
|
+
"endTimeUnixNano": str(span.end_time_unix_nano),
|
|
264
|
+
"attributes": _kv_json(span.attributes),
|
|
265
|
+
"events": [
|
|
266
|
+
{
|
|
267
|
+
"name": e.name,
|
|
268
|
+
"timeUnixNano": str(e.time_unix_nano),
|
|
269
|
+
"attributes": _kv_json(e.attributes),
|
|
270
|
+
}
|
|
271
|
+
for e in span.events
|
|
272
|
+
],
|
|
273
|
+
"links": [
|
|
274
|
+
{
|
|
275
|
+
"traceId": link.trace_id,
|
|
276
|
+
"spanId": link.span_id,
|
|
277
|
+
"attributes": _kv_json(link.attributes),
|
|
278
|
+
}
|
|
279
|
+
for link in span.links
|
|
280
|
+
],
|
|
281
|
+
"status": {"code": _STATUS_NUMBERS[span.status_code], "message": span.status_message},
|
|
282
|
+
}
|
|
283
|
+
if span.parent_span_id:
|
|
284
|
+
result["parentSpanId"] = span.parent_span_id
|
|
285
|
+
return result
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def _kv_json(attributes: dict[str, Any]) -> list[dict[str, Any]]:
|
|
289
|
+
return [{"key": key, "value": _any_value_json(value)} for key, value in attributes.items()]
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def _any_value_json(value: Any) -> dict[str, Any]:
|
|
293
|
+
if value is None:
|
|
294
|
+
return {}
|
|
295
|
+
if isinstance(value, bool): # before int: bool is a subclass of int
|
|
296
|
+
return {"boolValue": value}
|
|
297
|
+
if isinstance(value, int):
|
|
298
|
+
return {"intValue": str(value)}
|
|
299
|
+
if isinstance(value, float):
|
|
300
|
+
return {"doubleValue": value}
|
|
301
|
+
if isinstance(value, list):
|
|
302
|
+
return {"arrayValue": {"values": [_any_value_json(v) for v in value]}}
|
|
303
|
+
if isinstance(value, dict):
|
|
304
|
+
return {"kvlistValue": {"values": _kv_json(value)}}
|
|
305
|
+
return {"stringValue": str(value)}
|
loopview/ingest/raw.py
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""RawSpan: an OpenTelemetry span decoded from OTLP, with no interpretation.
|
|
2
|
+
|
|
3
|
+
This is the boundary between "wire format" and "meaning". The ingest layer turns
|
|
4
|
+
OTLP bytes into RawSpans; the normalize layer (M3) turns RawSpans into the
|
|
5
|
+
internal schema. Nothing here knows about gen_ai, OpenInference or any framework.
|
|
6
|
+
|
|
7
|
+
Attribute values are plain Python values (str, bool, int, float, list, dict), so
|
|
8
|
+
RawSpans are easy to test, print and write to JSON.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from typing import Any, Literal
|
|
12
|
+
|
|
13
|
+
from pydantic import BaseModel, Field
|
|
14
|
+
|
|
15
|
+
SpanKind = Literal["unspecified", "internal", "server", "client", "producer", "consumer"]
|
|
16
|
+
StatusCode = Literal["unset", "ok", "error"]
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class SpanEvent(BaseModel):
|
|
20
|
+
name: str
|
|
21
|
+
time_unix_nano: int
|
|
22
|
+
attributes: dict[str, Any] = Field(default_factory=dict)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class SpanLink(BaseModel):
|
|
26
|
+
trace_id: str
|
|
27
|
+
span_id: str
|
|
28
|
+
attributes: dict[str, Any] = Field(default_factory=dict)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class RawSpan(BaseModel):
|
|
32
|
+
trace_id: str # 32 lowercase hex chars
|
|
33
|
+
span_id: str # 16 lowercase hex chars
|
|
34
|
+
parent_span_id: str | None = None # None for a root span
|
|
35
|
+
name: str
|
|
36
|
+
kind: SpanKind = "unspecified"
|
|
37
|
+
start_time_unix_nano: int
|
|
38
|
+
end_time_unix_nano: int
|
|
39
|
+
attributes: dict[str, Any] = Field(default_factory=dict)
|
|
40
|
+
events: list[SpanEvent] = Field(default_factory=list)
|
|
41
|
+
links: list[SpanLink] = Field(default_factory=list)
|
|
42
|
+
status_code: StatusCode = "unset"
|
|
43
|
+
status_message: str = ""
|
|
44
|
+
# Where the span came from: the process (resource) and the library (scope).
|
|
45
|
+
resource_attributes: dict[str, Any] = Field(default_factory=dict)
|
|
46
|
+
scope_name: str = ""
|
|
47
|
+
scope_version: str = ""
|