loopview 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. loopview/__init__.py +3 -0
  2. loopview/app.py +252 -0
  3. loopview/cli.py +128 -0
  4. loopview/cost/__init__.py +1 -0
  5. loopview/cost/pricing.json +62 -0
  6. loopview/cost/pricing.py +74 -0
  7. loopview/cost/split.py +228 -0
  8. loopview/demo_data/flagship.otlp.jsonl +13 -0
  9. loopview/devtools/__init__.py +1 -0
  10. loopview/devtools/dump_normalized.py +211 -0
  11. loopview/ingest/__init__.py +1 -0
  12. loopview/ingest/otlp.py +305 -0
  13. loopview/ingest/raw.py +47 -0
  14. loopview/live.py +154 -0
  15. loopview/normalize/__init__.py +1 -0
  16. loopview/normalize/adapters/__init__.py +14 -0
  17. loopview/normalize/adapters/base.py +87 -0
  18. loopview/normalize/adapters/gen_ai.py +252 -0
  19. loopview/normalize/adapters/generic.py +22 -0
  20. loopview/normalize/adapters/openinference.py +367 -0
  21. loopview/normalize/derived_tools.py +90 -0
  22. loopview/normalize/loop_nodes.py +193 -0
  23. loopview/normalize/normalizer.py +326 -0
  24. loopview/normalize/schema.py +161 -0
  25. loopview/normalize/transitions.py +106 -0
  26. loopview/py.typed +0 -0
  27. loopview/static/assets/elk-worker.min-DfmSo98M.js +22 -0
  28. loopview/static/assets/index-CEGppBgk.css +1 -0
  29. loopview/static/assets/index-DzC96LiQ.js +21 -0
  30. loopview/static/assets/inter-cyrillic-ext-wght-normal-BOeWTOD4.woff2 +0 -0
  31. loopview/static/assets/inter-cyrillic-wght-normal-DqGufNeO.woff2 +0 -0
  32. loopview/static/assets/inter-greek-ext-wght-normal-DlzME5K_.woff2 +0 -0
  33. loopview/static/assets/inter-greek-wght-normal-CkhJZR-_.woff2 +0 -0
  34. loopview/static/assets/inter-latin-ext-wght-normal-DO1Apj_S.woff2 +0 -0
  35. loopview/static/assets/inter-latin-wght-normal-Dx4kXJAl.woff2 +0 -0
  36. loopview/static/assets/inter-vietnamese-wght-normal-CBcvBZtf.woff2 +0 -0
  37. loopview/static/assets/jetbrains-mono-cyrillic-wght-normal-D73BlboJ.woff2 +0 -0
  38. loopview/static/assets/jetbrains-mono-greek-wght-normal-Bw9x6K1M.woff2 +0 -0
  39. loopview/static/assets/jetbrains-mono-latin-ext-wght-normal-DBQx-q_a.woff2 +0 -0
  40. loopview/static/assets/jetbrains-mono-latin-wght-normal-B9CIFXIH.woff2 +0 -0
  41. loopview/static/assets/jetbrains-mono-vietnamese-wght-normal-Bt-aOZkq.woff2 +0 -0
  42. loopview/static/index.html +13 -0
  43. loopview/store/__init__.py +1 -0
  44. loopview/store/capture.py +112 -0
  45. loopview/store/memory.py +165 -0
  46. loopview/tools/__init__.py +1 -0
  47. loopview/tools/report.py +347 -0
  48. loopview-0.1.0.dist-info/METADATA +72 -0
  49. loopview-0.1.0.dist-info/RECORD +51 -0
  50. loopview-0.1.0.dist-info/WHEEL +4 -0
  51. loopview-0.1.0.dist-info/entry_points.txt +3 -0
@@ -0,0 +1 @@
1
+ """Developer tools, not used at runtime."""
@@ -0,0 +1,211 @@
1
+ """Write the normalized form of every fixture as JSON.
2
+
3
+ Used for the UI tests, and for the hosted demo (a static build of the UI that
4
+ reads recorded runs from files instead of a server). Also writes tools.json, the
5
+ Tools tab's report.
6
+
7
+ Most fixtures hold one run and become `<name>.json`. A fixture with several runs
8
+ (the MCP tools study) becomes `<name>-01.json`, `<name>-02.json`... but only for
9
+ the demo (--index): the UI tests don't use them, and they are several MB.
10
+
11
+ With --index (the demo), only the fixtures in DEMO_RUNS and DEMO_TOOLS_ONLY are
12
+ written. index.json lists each run file with its RunInfo, so the demo can show
13
+ the run list without downloading every run. DEMO_RUNS are the curated examples
14
+ shown in the run list, each with a short description of the agent and its task;
15
+ DEMO_TOOLS_ONLY runs are not listed but feed the Tools tab, whose links can still
16
+ open them. Without --index, tools.json covers every fixture.
17
+
18
+ Usage (from server/):
19
+ uv run python -m loopview.devtools.dump_normalized ../ui/src/test-data
20
+ uv run python -m loopview.devtools.dump_normalized ../ui/pages-public/demo --index
21
+ """
22
+
23
+ import json
24
+ import sys
25
+ from pathlib import Path
26
+ from typing import Any
27
+
28
+ from loopview.normalize.normalizer import normalize_run
29
+ from loopview.normalize.schema import NormalizedRun
30
+ from loopview.store.capture import load_capture
31
+ from loopview.store.memory import TraceStore
32
+ from loopview.tools.report import tools_report
33
+
34
+ REPO = Path(__file__).resolve().parents[4]
35
+ FIXTURES = REPO / "fixtures"
36
+ MCP_STUDY_TASKS = REPO / "examples" / "mcp_tools_study_tasks.txt"
37
+
38
+ # The examples the hosted demo lists, in order, with what the demo says about each.
39
+ # `prompt` is the task the agent was given, as written in the example's source.
40
+ DEMO_RUNS: dict[str, dict[str, Any]] = {
41
+ "flagship": {
42
+ "title": "Database comparison team",
43
+ "framework": "LangGraph",
44
+ "source": "examples/demo/flagship.py",
45
+ "summary": (
46
+ "A supervisor splits the question into three briefs. Three analyst agents research "
47
+ "benchmarks, operations and ecosystem in parallel, each with its own tools. Their "
48
+ "notes are merged, a critic sends the draft back once, and a writer produces the "
49
+ "answer."
50
+ ),
51
+ "prompt": (
52
+ "Compare PostgreSQL, SQLite and DuckDB for a small internal analytics dashboard: "
53
+ "5 users, about 20 GB of event data, loaded once a day. Recommend one."
54
+ ),
55
+ "look_for": [
56
+ "the fan-out to three analysts running at the same time",
57
+ "fetch_repo_stats failing on its first call, then the retry",
58
+ "the critic's loop back to synthesize, then the handoff to the writer",
59
+ "the Cost tab: a 4,300-token handbook read from the prompt cache",
60
+ ],
61
+ },
62
+ "multi_agent_pydantic": {
63
+ "title": "Laptop advice, multi-agent",
64
+ "framework": "Pydantic AI",
65
+ "source": "examples/multi_agent_pydantic.py",
66
+ "summary": (
67
+ "A coordinator agent delegates to two researchers (specs and reviews) that run in "
68
+ "parallel behind one tool call, then hands its notes to a writer agent."
69
+ ),
70
+ "prompt": (
71
+ "A student needs a laptop for programming and light video editing, budget 1300 EUR. "
72
+ "Compare the Aster Pro 14 and the Kestrel Air 15 and recommend one."
73
+ ),
74
+ "look_for": [
75
+ "agents nested inside a tool call (delegation)",
76
+ "the two researchers overlapping in time",
77
+ "the handoff from the coordinator to the writer",
78
+ ],
79
+ },
80
+ "langgraph_router": {
81
+ "title": "Date question router",
82
+ "framework": "LangGraph",
83
+ "source": "examples/langgraph_router.py",
84
+ "summary": (
85
+ "A classifier routes the question to a tool-using agent. The agent loops with its date "
86
+ "tools, then a strict reviewer asks for one revision and sends it back."
87
+ ),
88
+ "prompt": (
89
+ "How many days are there from 2026-09-30 until the next February 29th, "
90
+ "and what day of the week will that February 29th be?"
91
+ ),
92
+ "look_for": [
93
+ "the conditional route out of classify",
94
+ "the agent and tools loop",
95
+ "the reviewer's loop back to an earlier node",
96
+ ],
97
+ },
98
+ "failing_tools_pydantic": {
99
+ "title": "Shop assistant, failing tools",
100
+ "framework": "Pydantic AI + MCP",
101
+ "source": "examples/failing_tools_pydantic.py",
102
+ "summary": (
103
+ "A support agent with an order tool and an inventory MCP server. The question is "
104
+ "worded so both tools fail first: the agent has to read the errors and recover."
105
+ ),
106
+ "prompt": (
107
+ "What is the status of order 1234, and how many units of SKU-0099 (the blue mug) "
108
+ "are in stock?"
109
+ ),
110
+ "look_for": [
111
+ "red tool calls: a ModelRetry and an MCP isError result",
112
+ "what the model does next: fixes its arguments, or switches tool",
113
+ "the Tools tab, where these recoveries are counted",
114
+ ],
115
+ },
116
+ "react_anthropic": {
117
+ "title": "Trip budget, hand-rolled loop",
118
+ "framework": "Anthropic SDK",
119
+ "source": "examples/react_anthropic.py",
120
+ "summary": (
121
+ "A plain ReAct loop written by hand on the Anthropic SDK, traced with the "
122
+ "OpenTelemetry GenAI conventions. Extended thinking is on, so the model's reasoning "
123
+ "is recorded."
124
+ ),
125
+ "prompt": (
126
+ "I'm planning 3 nights in Lisbon and 2 nights in Porto. Look up the average "
127
+ "hotel price per night in each city, then tell me the total in EUR and in USD."
128
+ ),
129
+ "look_for": [
130
+ "the model's thinking, shown inside each model call",
131
+ "several tool calls asked for in one model turn",
132
+ ],
133
+ },
134
+ }
135
+ # Not listed (20 short tasks against GitHub's MCP server), but the Tools tab is built on them.
136
+ DEMO_TOOLS_ONLY = ["mcp_tools_study"]
137
+
138
+
139
+ def tools_only_about(name: str, run: NormalizedRun) -> dict[str, Any] | None:
140
+ """What the demo says about an unlisted run that a Tools tab link opened."""
141
+ if name != "mcp_tools_study" or not run.run.name.startswith("task "):
142
+ return None # the study's first run only opens the MCP connection
143
+ lines = MCP_STUDY_TASKS.read_text(encoding="utf-8").splitlines()
144
+ tasks = [line for line in lines if line.strip() and not line.startswith("#")]
145
+ number = int(run.run.name.removeprefix("task "))
146
+ return {
147
+ "title": f"GitHub MCP study, task {number}",
148
+ "framework": "Pydantic AI + GitHub MCP",
149
+ "source": "examples/mcp_tools_study.py",
150
+ "summary": (
151
+ "One of 20 read-only questions a Pydantic AI agent answered with the tools of GitHub's "
152
+ "official MCP server. Together, these runs feed the Tools tab."
153
+ ),
154
+ "prompt": tasks[number - 1],
155
+ "look_for": [
156
+ "which of the server's many tools the model picks",
157
+ "how big the tool results are, in the Cost tab",
158
+ ],
159
+ }
160
+
161
+
162
+ def main() -> None:
163
+ out = Path(sys.argv[1])
164
+ demo = "--index" in sys.argv
165
+ out.mkdir(parents=True, exist_ok=True)
166
+ all_runs: list[NormalizedRun] = []
167
+ index: list[tuple[int, str, NormalizedRun, str]] = [] # (demo position, file, run, fixture)
168
+ for fixture in sorted(FIXTURES.glob("*.otlp.jsonl")):
169
+ name = fixture.name.removesuffix(".otlp.jsonl")
170
+ if demo and name not in DEMO_RUNS and name not in DEMO_TOOLS_ONLY:
171
+ continue
172
+ store = TraceStore()
173
+ load_capture(store, fixture)
174
+ runs = [
175
+ normalize_run(run, now_ns=run.last_received_ns)
176
+ for run in sorted(store.runs(), key=lambda r: r.first_received_ns)
177
+ ]
178
+ all_runs += runs
179
+ if len(runs) > 1 and not demo:
180
+ print(f"skipped {name} ({len(runs)} runs: written for the demo only)")
181
+ continue
182
+ position = list(DEMO_RUNS).index(name) if name in DEMO_RUNS else len(DEMO_RUNS)
183
+ for i, normalized in enumerate(runs, 1):
184
+ file = f"{name}.json" if len(runs) == 1 else f"{name}-{i:02d}.json"
185
+ (out / file).write_text(normalized.model_dump_json(), encoding="utf-8")
186
+ index.append((position, file, normalized, name))
187
+ print(f"wrote {name} ({len(runs)} run{'s' if len(runs) > 1 else ''})")
188
+ (out / "tools.json").write_text(json.dumps(tools_report(all_runs)), encoding="utf-8")
189
+ print("wrote tools.json")
190
+ if demo:
191
+ missing = [
192
+ name for name in [*DEMO_RUNS, *DEMO_TOOLS_ONLY] if not any(e[3] == name for e in index)
193
+ ]
194
+ if missing:
195
+ sys.exit(f"no fixture for {', '.join(missing)}")
196
+ index.sort(key=lambda entry: entry[0]) # stable: a fixture's runs keep their order
197
+ entries = [
198
+ {
199
+ "file": file,
200
+ "run": n.run.model_dump(mode="json"),
201
+ "listed": name in DEMO_RUNS,
202
+ "about": DEMO_RUNS.get(name) or tools_only_about(name, n),
203
+ }
204
+ for _, file, n, name in index
205
+ ]
206
+ (out / "index.json").write_text(json.dumps(entries), encoding="utf-8")
207
+ print("wrote index.json")
208
+
209
+
210
+ if __name__ == "__main__":
211
+ main()
@@ -0,0 +1 @@
1
+ """Ingestion: OTLP/HTTP request bodies in, RawSpans out."""
@@ -0,0 +1,305 @@
1
+ """Decode OTLP/HTTP trace export requests into RawSpans.
2
+
3
+ Follows the OTLP specification v1.11.0 (https://opentelemetry.io/docs/specs/otlp/).
4
+
5
+ Both encodings go through one path: JSON is parsed into the same protobuf message
6
+ as the binary encoding, then a single function converts the message to RawSpans.
7
+ That works because OTLP/JSON is the standard protobuf JSON mapping with one
8
+ exception: trace and span ids are hex strings instead of base64. We rewrite those
9
+ ids before parsing and let protobuf handle everything else (camelCase keys, enums
10
+ as integers, 64-bit integers as strings, unknown fields ignored).
11
+ """
12
+
13
+ import base64
14
+ import json
15
+ import zlib
16
+ from typing import Any
17
+
18
+ from google.protobuf import json_format
19
+ from opentelemetry.proto.collector.trace.v1.trace_service_pb2 import (
20
+ ExportTraceServiceRequest,
21
+ ExportTraceServiceResponse,
22
+ )
23
+ from opentelemetry.proto.common.v1.common_pb2 import AnyValue, KeyValue
24
+ from opentelemetry.proto.trace.v1.trace_pb2 import Span, Status
25
+
26
+ from loopview.ingest.raw import RawSpan, SpanEvent, SpanKind, SpanLink, StatusCode
27
+
28
+ PROTOBUF = "application/x-protobuf"
29
+ JSON = "application/json"
30
+
31
+ # Largest request body we accept, after decompression. Big enough for thousands of
32
+ # spans with full prompts; small enough that a bad request can't eat all memory.
33
+ MAX_BODY_BYTES = 64 * 1024 * 1024
34
+
35
+
36
+ class OtlpDecodeError(ValueError):
37
+ """The request could not be decoded. Maps to HTTP 400 (not retryable)."""
38
+
39
+
40
+ class UnsupportedContentType(OtlpDecodeError):
41
+ """Neither protobuf nor JSON. Maps to HTTP 415."""
42
+
43
+
44
+ class BodyTooLarge(OtlpDecodeError):
45
+ """Maps to HTTP 413 (not retryable)."""
46
+
47
+
48
+ def decode_request(
49
+ body: bytes, content_type: str | None, content_encoding: str | None = None
50
+ ) -> list[RawSpan]:
51
+ """Decode one POST /v1/traces body into RawSpans."""
52
+ return decode_body(decompress(body, content_encoding), content_type)
53
+
54
+
55
+ def decode_body(body: bytes, content_type: str | None) -> list[RawSpan]:
56
+ """Decode an already decompressed body into RawSpans."""
57
+ media_type = (content_type or "").split(";")[0].strip().lower()
58
+ if media_type == PROTOBUF:
59
+ request = _parse_protobuf(body)
60
+ elif media_type == JSON:
61
+ request = _parse_json(body)
62
+ else:
63
+ raise UnsupportedContentType(f"unsupported content type: {content_type!r}")
64
+ return request_to_spans(request)
65
+
66
+
67
+ def encode_success_response(content_type: str | None) -> tuple[bytes, str]:
68
+ """An empty ExportTraceServiceResponse in the same encoding as the request."""
69
+ response = ExportTraceServiceResponse()
70
+ if (content_type or "").lower().startswith(JSON):
71
+ return json_format.MessageToJson(response).encode(), JSON
72
+ return response.SerializeToString(), PROTOBUF
73
+
74
+
75
+ def request_to_spans(request: ExportTraceServiceRequest) -> list[RawSpan]:
76
+ spans: list[RawSpan] = []
77
+ for resource_spans in request.resource_spans:
78
+ resource_attributes = _attributes(resource_spans.resource.attributes)
79
+ for scope_spans in resource_spans.scope_spans:
80
+ scope = scope_spans.scope
81
+ for span in scope_spans.spans:
82
+ spans.append(_span(span, resource_attributes, scope.name, scope.version))
83
+ return spans
84
+
85
+
86
+ # --- decoding the body ----------------------------------------------------------
87
+
88
+
89
+ def decompress(body: bytes, content_encoding: str | None) -> bytes:
90
+ """Undo Content-Encoding (identity or gzip), enforcing MAX_BODY_BYTES."""
91
+ encoding = (content_encoding or "identity").strip().lower()
92
+ if encoding == "identity":
93
+ result = body
94
+ elif encoding == "gzip":
95
+ # wbits=16+MAX_WBITS means "expect a gzip header". max_length stops a small
96
+ # compressed body from expanding into gigabytes.
97
+ decompressor = zlib.decompressobj(16 + zlib.MAX_WBITS)
98
+ try:
99
+ result = decompressor.decompress(body, MAX_BODY_BYTES + 1)
100
+ except zlib.error as exc:
101
+ raise OtlpDecodeError(f"invalid gzip body: {exc}") from exc
102
+ else:
103
+ raise OtlpDecodeError(f"unsupported content encoding: {content_encoding!r}")
104
+ if len(result) > MAX_BODY_BYTES:
105
+ raise BodyTooLarge(f"request body larger than {MAX_BODY_BYTES} bytes")
106
+ return result
107
+
108
+
109
+ def _parse_protobuf(body: bytes) -> ExportTraceServiceRequest:
110
+ request = ExportTraceServiceRequest()
111
+ try:
112
+ request.ParseFromString(body)
113
+ except Exception as exc: # protobuf raises DecodeError, but be safe
114
+ raise OtlpDecodeError(f"invalid protobuf body: {exc}") from exc
115
+ return request
116
+
117
+
118
+ def _parse_json(body: bytes) -> ExportTraceServiceRequest:
119
+ try:
120
+ data = json.loads(body)
121
+ _hex_ids_to_base64(data)
122
+ return json_format.ParseDict(data, ExportTraceServiceRequest(), ignore_unknown_fields=True)
123
+ except (ValueError, TypeError, json_format.ParseError) as exc:
124
+ raise OtlpDecodeError(f"invalid JSON body: {exc}") from exc
125
+
126
+
127
+ # OTLP/JSON fields that hold hex ids. Protobuf's JSON parser expects base64 for bytes.
128
+ _ID_FIELDS = ("traceId", "spanId", "parentSpanId")
129
+
130
+
131
+ def _hex_ids_to_base64(data: Any) -> None:
132
+ """Rewrite hex trace/span ids in place, on spans and on span links."""
133
+ for resource_spans in data.get("resourceSpans", []):
134
+ for scope_spans in resource_spans.get("scopeSpans", []):
135
+ for span in scope_spans.get("spans", []):
136
+ _convert_ids(span)
137
+ for link in span.get("links", []):
138
+ _convert_ids(link)
139
+
140
+
141
+ def _convert_ids(obj: dict[str, Any]) -> None:
142
+ for field in _ID_FIELDS:
143
+ value = obj.get(field)
144
+ if value:
145
+ obj[field] = base64.b64encode(bytes.fromhex(value)).decode()
146
+
147
+
148
+ # --- protobuf message to RawSpan ------------------------------------------------
149
+
150
+ _KINDS: dict[int, SpanKind] = {
151
+ Span.SPAN_KIND_UNSPECIFIED: "unspecified",
152
+ Span.SPAN_KIND_INTERNAL: "internal",
153
+ Span.SPAN_KIND_SERVER: "server",
154
+ Span.SPAN_KIND_CLIENT: "client",
155
+ Span.SPAN_KIND_PRODUCER: "producer",
156
+ Span.SPAN_KIND_CONSUMER: "consumer",
157
+ }
158
+
159
+ _STATUS: dict[int, StatusCode] = {
160
+ Status.STATUS_CODE_UNSET: "unset",
161
+ Status.STATUS_CODE_OK: "ok",
162
+ Status.STATUS_CODE_ERROR: "error",
163
+ }
164
+
165
+
166
+ def _span(
167
+ span: Span, resource_attributes: dict[str, Any], scope_name: str, scope_version: str
168
+ ) -> RawSpan:
169
+ return RawSpan(
170
+ trace_id=span.trace_id.hex(),
171
+ span_id=span.span_id.hex(),
172
+ parent_span_id=span.parent_span_id.hex() or None,
173
+ name=span.name,
174
+ kind=_KINDS.get(span.kind, "unspecified"),
175
+ start_time_unix_nano=span.start_time_unix_nano,
176
+ end_time_unix_nano=span.end_time_unix_nano,
177
+ attributes=_attributes(span.attributes),
178
+ events=[
179
+ SpanEvent(
180
+ name=event.name,
181
+ time_unix_nano=event.time_unix_nano,
182
+ attributes=_attributes(event.attributes),
183
+ )
184
+ for event in span.events
185
+ ],
186
+ links=[
187
+ SpanLink(
188
+ trace_id=link.trace_id.hex(),
189
+ span_id=link.span_id.hex(),
190
+ attributes=_attributes(link.attributes),
191
+ )
192
+ for link in span.links
193
+ ],
194
+ status_code=_STATUS.get(span.status.code, "unset"),
195
+ status_message=span.status.message,
196
+ resource_attributes=resource_attributes,
197
+ scope_name=scope_name,
198
+ scope_version=scope_version,
199
+ )
200
+
201
+
202
+ def _attributes(key_values: "list[KeyValue] | Any") -> dict[str, Any]:
203
+ return {kv.key: _value(kv.value) for kv in key_values}
204
+
205
+
206
+ def _value(value: AnyValue) -> Any:
207
+ """Convert an OTLP AnyValue into a plain Python value."""
208
+ kind = value.WhichOneof("value")
209
+ if kind is None:
210
+ return None
211
+ if kind == "array_value":
212
+ return [_value(v) for v in value.array_value.values]
213
+ if kind == "kvlist_value":
214
+ return _attributes(value.kvlist_value.values)
215
+ if kind == "bytes_value":
216
+ # Keep RawSpans JSON-friendly: bytes become base64 text.
217
+ return base64.b64encode(value.bytes_value).decode()
218
+ return getattr(value, kind) # string_value, bool_value, int_value, double_value
219
+
220
+
221
+ # --- RawSpan back to OTLP/JSON (for exporting a run) ----------------------------------
222
+
223
+ _KIND_NUMBERS = {name: number for number, name in _KINDS.items()}
224
+ _STATUS_NUMBERS = {name: number for number, name in _STATUS.items()}
225
+
226
+
227
+ def spans_to_otlp_json(spans: list[RawSpan]) -> dict[str, Any]:
228
+ """Encode RawSpans as an OTLP/JSON ExportTraceServiceRequest.
229
+
230
+ The inverse of decoding, so an exported run can be imported again through
231
+ the normal receiver path. (Bytes attributes come back as base64 strings.)"""
232
+ groups: dict[tuple[str, str, str], list[RawSpan]] = {}
233
+ for span in spans:
234
+ key = (
235
+ json.dumps(span.resource_attributes, sort_keys=True),
236
+ span.scope_name,
237
+ span.scope_version,
238
+ )
239
+ groups.setdefault(key, []).append(span)
240
+ resource_spans = []
241
+ for (resource_json, scope_name, scope_version), group in groups.items():
242
+ resource_spans.append(
243
+ {
244
+ "resource": {"attributes": _kv_json(json.loads(resource_json))},
245
+ "scopeSpans": [
246
+ {
247
+ "scope": {"name": scope_name, "version": scope_version},
248
+ "spans": [_span_json(s) for s in group],
249
+ }
250
+ ],
251
+ }
252
+ )
253
+ return {"resourceSpans": resource_spans}
254
+
255
+
256
+ def _span_json(span: RawSpan) -> dict[str, Any]:
257
+ result: dict[str, Any] = {
258
+ "traceId": span.trace_id,
259
+ "spanId": span.span_id,
260
+ "name": span.name,
261
+ "kind": _KIND_NUMBERS[span.kind],
262
+ "startTimeUnixNano": str(span.start_time_unix_nano),
263
+ "endTimeUnixNano": str(span.end_time_unix_nano),
264
+ "attributes": _kv_json(span.attributes),
265
+ "events": [
266
+ {
267
+ "name": e.name,
268
+ "timeUnixNano": str(e.time_unix_nano),
269
+ "attributes": _kv_json(e.attributes),
270
+ }
271
+ for e in span.events
272
+ ],
273
+ "links": [
274
+ {
275
+ "traceId": link.trace_id,
276
+ "spanId": link.span_id,
277
+ "attributes": _kv_json(link.attributes),
278
+ }
279
+ for link in span.links
280
+ ],
281
+ "status": {"code": _STATUS_NUMBERS[span.status_code], "message": span.status_message},
282
+ }
283
+ if span.parent_span_id:
284
+ result["parentSpanId"] = span.parent_span_id
285
+ return result
286
+
287
+
288
+ def _kv_json(attributes: dict[str, Any]) -> list[dict[str, Any]]:
289
+ return [{"key": key, "value": _any_value_json(value)} for key, value in attributes.items()]
290
+
291
+
292
+ def _any_value_json(value: Any) -> dict[str, Any]:
293
+ if value is None:
294
+ return {}
295
+ if isinstance(value, bool): # before int: bool is a subclass of int
296
+ return {"boolValue": value}
297
+ if isinstance(value, int):
298
+ return {"intValue": str(value)}
299
+ if isinstance(value, float):
300
+ return {"doubleValue": value}
301
+ if isinstance(value, list):
302
+ return {"arrayValue": {"values": [_any_value_json(v) for v in value]}}
303
+ if isinstance(value, dict):
304
+ return {"kvlistValue": {"values": _kv_json(value)}}
305
+ return {"stringValue": str(value)}
loopview/ingest/raw.py ADDED
@@ -0,0 +1,47 @@
1
+ """RawSpan: an OpenTelemetry span decoded from OTLP, with no interpretation.
2
+
3
+ This is the boundary between "wire format" and "meaning". The ingest layer turns
4
+ OTLP bytes into RawSpans; the normalize layer (M3) turns RawSpans into the
5
+ internal schema. Nothing here knows about gen_ai, OpenInference or any framework.
6
+
7
+ Attribute values are plain Python values (str, bool, int, float, list, dict), so
8
+ RawSpans are easy to test, print and write to JSON.
9
+ """
10
+
11
+ from typing import Any, Literal
12
+
13
+ from pydantic import BaseModel, Field
14
+
15
+ SpanKind = Literal["unspecified", "internal", "server", "client", "producer", "consumer"]
16
+ StatusCode = Literal["unset", "ok", "error"]
17
+
18
+
19
+ class SpanEvent(BaseModel):
20
+ name: str
21
+ time_unix_nano: int
22
+ attributes: dict[str, Any] = Field(default_factory=dict)
23
+
24
+
25
+ class SpanLink(BaseModel):
26
+ trace_id: str
27
+ span_id: str
28
+ attributes: dict[str, Any] = Field(default_factory=dict)
29
+
30
+
31
+ class RawSpan(BaseModel):
32
+ trace_id: str # 32 lowercase hex chars
33
+ span_id: str # 16 lowercase hex chars
34
+ parent_span_id: str | None = None # None for a root span
35
+ name: str
36
+ kind: SpanKind = "unspecified"
37
+ start_time_unix_nano: int
38
+ end_time_unix_nano: int
39
+ attributes: dict[str, Any] = Field(default_factory=dict)
40
+ events: list[SpanEvent] = Field(default_factory=list)
41
+ links: list[SpanLink] = Field(default_factory=list)
42
+ status_code: StatusCode = "unset"
43
+ status_message: str = ""
44
+ # Where the span came from: the process (resource) and the library (scope).
45
+ resource_attributes: dict[str, Any] = Field(default_factory=dict)
46
+ scope_name: str = ""
47
+ scope_version: str = ""