traceburn 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- traceburn/__init__.py +62 -0
- traceburn/analyze/__init__.py +3 -0
- traceburn/analyze/diff.py +243 -0
- traceburn/analyze/flamegraph.py +113 -0
- traceburn/analyze/replay.py +351 -0
- traceburn/analyze/waste/__init__.py +88 -0
- traceburn/analyze/waste/_common.py +158 -0
- traceburn/analyze/waste/cache.py +126 -0
- traceburn/analyze/waste/context_bloat.py +111 -0
- traceburn/analyze/waste/duplicates.py +178 -0
- traceburn/analyze/waste/loops.py +121 -0
- traceburn/analyze/waste/model_overkill.py +94 -0
- traceburn/cli.py +336 -0
- traceburn/instrument/__init__.py +47 -0
- traceburn/instrument/_util.py +260 -0
- traceburn/instrument/anthropic.py +428 -0
- traceburn/instrument/openai.py +458 -0
- traceburn/pricing.json +290 -0
- traceburn/pricing.py +180 -0
- traceburn/recorder.py +372 -0
- traceburn/schema.py +158 -0
- traceburn/store.py +393 -0
- traceburn/ui/__init__.py +5 -0
- traceburn/ui/server.py +172 -0
- traceburn/ui/static/app.js +454 -0
- traceburn/ui/static/index.html +48 -0
- traceburn/ui/static/style.css +154 -0
- traceburn-0.1.0.dist-info/METADATA +259 -0
- traceburn-0.1.0.dist-info/RECORD +34 -0
- traceburn-0.1.0.dist-info/WHEEL +4 -0
- traceburn-0.1.0.dist-info/entry_points.txt +2 -0
- traceburn-0.1.0.dist-info/licenses/LICENSE +21 -0
- traceburn_autoinstall.pth +1 -0
- traceburn_autoinstall.py +22 -0
traceburn/__init__.py
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""traceburn: a local-first tracer and efficiency profiler for AI agents.
|
|
2
|
+
|
|
3
|
+
Record every LLM and tool call an agent makes into one local SQLite file,
|
|
4
|
+
then see where the time and money went. No account, no server, no telemetry.
|
|
5
|
+
|
|
6
|
+
Quick use:
|
|
7
|
+
|
|
8
|
+
import traceburn
|
|
9
|
+
from traceburn import trace, span, session
|
|
10
|
+
|
|
11
|
+
@trace("plan", kind="agent")
|
|
12
|
+
def plan(state): ...
|
|
13
|
+
|
|
14
|
+
with session("nightly-run"):
|
|
15
|
+
with span("retrieve", kind="retrieval"):
|
|
16
|
+
docs = retriever(query)
|
|
17
|
+
answer = plan(state)
|
|
18
|
+
|
|
19
|
+
Or zero-config: ``traceburn.install()`` patches whichever of the openai and
|
|
20
|
+
anthropic clients are installed, and every call they make is recorded.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from .instrument import install, uninstall
|
|
24
|
+
from .pricing import ModelPrice, PricingTable
|
|
25
|
+
from .recorder import (
|
|
26
|
+
Recorder,
|
|
27
|
+
SpanHandle,
|
|
28
|
+
configure,
|
|
29
|
+
current_session,
|
|
30
|
+
current_span,
|
|
31
|
+
get_recorder,
|
|
32
|
+
session,
|
|
33
|
+
span,
|
|
34
|
+
trace,
|
|
35
|
+
)
|
|
36
|
+
from .schema import SPAN_KINDS, Finding, Session, Span, Trace
|
|
37
|
+
from .store import Store
|
|
38
|
+
|
|
39
|
+
__version__ = "0.1.0"
|
|
40
|
+
|
|
41
|
+
__all__ = [
|
|
42
|
+
"SPAN_KINDS",
|
|
43
|
+
"Finding",
|
|
44
|
+
"ModelPrice",
|
|
45
|
+
"PricingTable",
|
|
46
|
+
"Recorder",
|
|
47
|
+
"Session",
|
|
48
|
+
"Span",
|
|
49
|
+
"SpanHandle",
|
|
50
|
+
"Store",
|
|
51
|
+
"Trace",
|
|
52
|
+
"__version__",
|
|
53
|
+
"configure",
|
|
54
|
+
"current_session",
|
|
55
|
+
"current_span",
|
|
56
|
+
"get_recorder",
|
|
57
|
+
"install",
|
|
58
|
+
"session",
|
|
59
|
+
"span",
|
|
60
|
+
"trace",
|
|
61
|
+
"uninstall",
|
|
62
|
+
]
|
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
"""Diff two recorded traces: structure, metrics, and prompt/response text.
|
|
2
|
+
|
|
3
|
+
Spans are aligned recursively: children of matched parents pair up by
|
|
4
|
+
(name, kind) in order of occurrence, so the second "chat gpt-5" under the
|
|
5
|
+
same parent in run A matches the second in run B. Unmatched spans are
|
|
6
|
+
reported as added or removed. Matched LLM spans additionally get a unified
|
|
7
|
+
text diff of their prompts and responses when they differ.
|
|
8
|
+
|
|
9
|
+
The result dict is a public JSON interface.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import difflib
|
|
15
|
+
import json
|
|
16
|
+
import math
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
from ..schema import Span
|
|
20
|
+
from ..store import Store
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _tree(spans: list[Span]) -> dict[str | None, list[Span]]:
|
|
24
|
+
ids = {s.span_id for s in spans}
|
|
25
|
+
children: dict[str | None, list[Span]] = {}
|
|
26
|
+
for span in spans:
|
|
27
|
+
parent = span.parent_id if span.parent_id in ids else None
|
|
28
|
+
children.setdefault(parent, []).append(span)
|
|
29
|
+
for kids in children.values():
|
|
30
|
+
kids.sort(key=lambda s: s.start_ns)
|
|
31
|
+
return children
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _metric(span: Span, key: str) -> float:
|
|
35
|
+
value = span.attributes.get(key)
|
|
36
|
+
try:
|
|
37
|
+
parsed = float(value) if value is not None else 0.0
|
|
38
|
+
except (TypeError, ValueError):
|
|
39
|
+
return 0.0
|
|
40
|
+
return parsed if math.isfinite(parsed) else 0.0
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _duration_ms(span: Span) -> float:
|
|
44
|
+
if span.end_ns is None:
|
|
45
|
+
return 0.0
|
|
46
|
+
return (span.end_ns - span.start_ns) / 1e6
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _as_text(payload: Any) -> str:
|
|
50
|
+
if payload is None:
|
|
51
|
+
return ""
|
|
52
|
+
if isinstance(payload, str):
|
|
53
|
+
return payload
|
|
54
|
+
return json.dumps(payload, indent=2, sort_keys=True, default=str)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _text_diff(label: str, a: Any, b: Any, context: int = 2) -> list[str] | None:
|
|
58
|
+
text_a, text_b = _as_text(a), _as_text(b)
|
|
59
|
+
if text_a == text_b:
|
|
60
|
+
return None
|
|
61
|
+
lines = list(
|
|
62
|
+
difflib.unified_diff(
|
|
63
|
+
text_a.splitlines(),
|
|
64
|
+
text_b.splitlines(),
|
|
65
|
+
fromfile=f"a/{label}",
|
|
66
|
+
tofile=f"b/{label}",
|
|
67
|
+
lineterm="",
|
|
68
|
+
n=context,
|
|
69
|
+
)
|
|
70
|
+
)
|
|
71
|
+
return lines or None
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _span_brief(span: Span) -> dict[str, Any]:
|
|
75
|
+
return {
|
|
76
|
+
"span_id": span.span_id,
|
|
77
|
+
"name": span.name,
|
|
78
|
+
"kind": span.kind,
|
|
79
|
+
"status": span.status,
|
|
80
|
+
"duration_ms": round(_duration_ms(span), 3),
|
|
81
|
+
"cost_usd": span.attributes.get("cost_usd"),
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _match_level(
|
|
86
|
+
a_spans: list[Span],
|
|
87
|
+
b_spans: list[Span],
|
|
88
|
+
tree_a: dict,
|
|
89
|
+
tree_b: dict,
|
|
90
|
+
matched: list,
|
|
91
|
+
added: list,
|
|
92
|
+
removed: list,
|
|
93
|
+
) -> None:
|
|
94
|
+
consumed_b: set[str] = set()
|
|
95
|
+
counters_a: dict[tuple, int] = {}
|
|
96
|
+
slots_b: dict[tuple, list[Span]] = {}
|
|
97
|
+
for span in b_spans:
|
|
98
|
+
slots_b.setdefault((span.name, span.kind), []).append(span)
|
|
99
|
+
|
|
100
|
+
for span_a in a_spans:
|
|
101
|
+
key = (span_a.name, span_a.kind)
|
|
102
|
+
index = counters_a.get(key, 0)
|
|
103
|
+
counters_a[key] = index + 1
|
|
104
|
+
candidates = slots_b.get(key, [])
|
|
105
|
+
if index < len(candidates):
|
|
106
|
+
span_b = candidates[index]
|
|
107
|
+
consumed_b.add(span_b.span_id)
|
|
108
|
+
matched.append((span_a, span_b))
|
|
109
|
+
_match_level(
|
|
110
|
+
tree_a.get(span_a.span_id, []),
|
|
111
|
+
tree_b.get(span_b.span_id, []),
|
|
112
|
+
tree_a,
|
|
113
|
+
tree_b,
|
|
114
|
+
matched,
|
|
115
|
+
added,
|
|
116
|
+
removed,
|
|
117
|
+
)
|
|
118
|
+
else:
|
|
119
|
+
removed.append(span_a)
|
|
120
|
+
for descendant in _descendants(span_a, tree_a):
|
|
121
|
+
removed.append(descendant)
|
|
122
|
+
|
|
123
|
+
for span_b in b_spans:
|
|
124
|
+
if span_b.span_id not in consumed_b:
|
|
125
|
+
added.append(span_b)
|
|
126
|
+
for descendant in _descendants(span_b, tree_b):
|
|
127
|
+
added.append(descendant)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _descendants(span: Span, tree: dict) -> list[Span]:
|
|
131
|
+
out = []
|
|
132
|
+
stack = list(tree.get(span.span_id, []))
|
|
133
|
+
while stack:
|
|
134
|
+
current = stack.pop()
|
|
135
|
+
out.append(current)
|
|
136
|
+
stack.extend(tree.get(current.span_id, []))
|
|
137
|
+
return out
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def diff_traces(
|
|
141
|
+
store_a: Store,
|
|
142
|
+
trace_id_a: str,
|
|
143
|
+
store_b: Store | None = None,
|
|
144
|
+
trace_id_b: str | None = None,
|
|
145
|
+
) -> dict[str, Any]:
|
|
146
|
+
"""Compare two traces. With one store, pass trace_id_b only."""
|
|
147
|
+
store_b = store_b or store_a
|
|
148
|
+
if trace_id_b is None:
|
|
149
|
+
raise ValueError("trace_id_b is required")
|
|
150
|
+
spans_a = store_a.get_spans(trace_id_a, hydrate=True, hydrate_keys=("request", "response"))
|
|
151
|
+
spans_b = store_b.get_spans(trace_id_b, hydrate=True, hydrate_keys=("request", "response"))
|
|
152
|
+
|
|
153
|
+
tree_a, tree_b = _tree(spans_a), _tree(spans_b)
|
|
154
|
+
matched_pairs: list[tuple[Span, Span]] = []
|
|
155
|
+
added: list[Span] = []
|
|
156
|
+
removed: list[Span] = []
|
|
157
|
+
roots_a, roots_b = tree_a.get(None, []), tree_b.get(None, [])
|
|
158
|
+
if len(roots_a) == 1 and len(roots_b) == 1:
|
|
159
|
+
# Comparing two runs implies their roots correspond, even when the
|
|
160
|
+
# run was renamed between recordings.
|
|
161
|
+
matched_pairs.append((roots_a[0], roots_b[0]))
|
|
162
|
+
_match_level(
|
|
163
|
+
tree_a.get(roots_a[0].span_id, []),
|
|
164
|
+
tree_b.get(roots_b[0].span_id, []),
|
|
165
|
+
tree_a, tree_b, matched_pairs, added, removed,
|
|
166
|
+
)
|
|
167
|
+
else:
|
|
168
|
+
_match_level(
|
|
169
|
+
roots_a, roots_b, tree_a, tree_b, matched_pairs, added, removed,
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
matched = []
|
|
173
|
+
for span_a, span_b in matched_pairs:
|
|
174
|
+
entry: dict[str, Any] = {
|
|
175
|
+
"name": span_a.name,
|
|
176
|
+
"kind": span_a.kind,
|
|
177
|
+
"a": _span_brief(span_a),
|
|
178
|
+
"b": _span_brief(span_b),
|
|
179
|
+
"deltas": {
|
|
180
|
+
"duration_ms": round(_duration_ms(span_b) - _duration_ms(span_a), 3),
|
|
181
|
+
"cost_usd": _metric(span_b, "cost_usd") - _metric(span_a, "cost_usd"),
|
|
182
|
+
"input_tokens": int(
|
|
183
|
+
_metric(span_b, "gen_ai.usage.input_tokens")
|
|
184
|
+
+ _metric(span_b, "cached_input_tokens")
|
|
185
|
+
+ _metric(span_b, "cache_write_tokens")
|
|
186
|
+
- _metric(span_a, "gen_ai.usage.input_tokens")
|
|
187
|
+
- _metric(span_a, "cached_input_tokens")
|
|
188
|
+
- _metric(span_a, "cache_write_tokens")
|
|
189
|
+
),
|
|
190
|
+
"output_tokens": int(
|
|
191
|
+
_metric(span_b, "gen_ai.usage.output_tokens")
|
|
192
|
+
- _metric(span_a, "gen_ai.usage.output_tokens")
|
|
193
|
+
),
|
|
194
|
+
},
|
|
195
|
+
}
|
|
196
|
+
if span_a.kind == "llm":
|
|
197
|
+
request_diff = _text_diff(
|
|
198
|
+
"request", span_a.attributes.get("request"), span_b.attributes.get("request")
|
|
199
|
+
)
|
|
200
|
+
response_diff = _text_diff(
|
|
201
|
+
"response", span_a.attributes.get("response"), span_b.attributes.get("response")
|
|
202
|
+
)
|
|
203
|
+
if request_diff:
|
|
204
|
+
entry["request_diff"] = request_diff
|
|
205
|
+
if response_diff:
|
|
206
|
+
entry["response_diff"] = response_diff
|
|
207
|
+
matched.append(entry)
|
|
208
|
+
|
|
209
|
+
def totals(spans: list[Span]) -> dict[str, Any]:
|
|
210
|
+
return {
|
|
211
|
+
"spans": len(spans),
|
|
212
|
+
"llm_calls": sum(1 for s in spans if s.kind == "llm"),
|
|
213
|
+
"errors": sum(1 for s in spans if s.status == "error"),
|
|
214
|
+
"duration_ms": round(
|
|
215
|
+
max((s.end_ns or s.start_ns) for s in spans) / 1e6
|
|
216
|
+
- min(s.start_ns for s in spans) / 1e6,
|
|
217
|
+
3,
|
|
218
|
+
)
|
|
219
|
+
if spans
|
|
220
|
+
else 0.0,
|
|
221
|
+
"cost_usd": sum(_metric(s, "cost_usd") for s in spans),
|
|
222
|
+
"input_tokens": int(
|
|
223
|
+
sum(
|
|
224
|
+
_metric(s, "gen_ai.usage.input_tokens")
|
|
225
|
+
+ _metric(s, "cached_input_tokens")
|
|
226
|
+
+ _metric(s, "cache_write_tokens")
|
|
227
|
+
for s in spans
|
|
228
|
+
)
|
|
229
|
+
),
|
|
230
|
+
"output_tokens": int(
|
|
231
|
+
sum(_metric(s, "gen_ai.usage.output_tokens") for s in spans)
|
|
232
|
+
),
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
return {
|
|
236
|
+
"trace_a": trace_id_a,
|
|
237
|
+
"trace_b": trace_id_b,
|
|
238
|
+
"totals_a": totals(spans_a),
|
|
239
|
+
"totals_b": totals(spans_b),
|
|
240
|
+
"matched": matched,
|
|
241
|
+
"added": [_span_brief(s) for s in added],
|
|
242
|
+
"removed": [_span_brief(s) for s in removed],
|
|
243
|
+
}
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
"""Fold a trace into a frame tree for flamegraph and waterfall rendering.
|
|
2
|
+
|
|
3
|
+
The fold output is a public JSON interface: a tree of frames, each with a
|
|
4
|
+
total ``value`` and a ``self_value``, where value is either wall-clock
|
|
5
|
+
nanoseconds (``weight="latency"``) or estimated dollars (``weight="cost"``).
|
|
6
|
+
Latency self-time is the span's duration minus its children's, clamped at
|
|
7
|
+
zero because concurrent children can legitimately overlap their parent.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import math
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
from ..schema import Span
|
|
16
|
+
|
|
17
|
+
WEIGHTS = ("latency", "cost")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _duration(span: Span) -> int:
|
|
21
|
+
if span.end_ns is None:
|
|
22
|
+
return 0
|
|
23
|
+
return max(0, span.end_ns - span.start_ns)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _cost(span: Span) -> float:
|
|
27
|
+
value = span.attributes.get("cost_usd")
|
|
28
|
+
try:
|
|
29
|
+
cost = float(value) if value is not None else 0.0
|
|
30
|
+
except (TypeError, ValueError):
|
|
31
|
+
return 0.0
|
|
32
|
+
# The store sanitizes non-finite floats to strings, which float() would
|
|
33
|
+
# happily parse back into nan; keep JSON output JSON-serializable.
|
|
34
|
+
return cost if math.isfinite(cost) else 0.0
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def fold(spans: list[Span], weight: str = "latency") -> dict[str, Any]:
|
|
38
|
+
"""Fold spans into a frame tree rooted at a synthetic trace frame."""
|
|
39
|
+
if weight not in WEIGHTS:
|
|
40
|
+
raise ValueError(f"weight must be one of {WEIGHTS}, got {weight!r}")
|
|
41
|
+
|
|
42
|
+
ids = {s.span_id for s in spans}
|
|
43
|
+
children: dict[str | None, list[Span]] = {}
|
|
44
|
+
for span in spans:
|
|
45
|
+
parent = span.parent_id if span.parent_id in ids else None
|
|
46
|
+
children.setdefault(parent, []).append(span)
|
|
47
|
+
|
|
48
|
+
def build(span: Span) -> dict[str, Any]:
|
|
49
|
+
kids = sorted(children.get(span.span_id, []), key=lambda s: s.start_ns)
|
|
50
|
+
child_frames = [build(k) for k in kids]
|
|
51
|
+
if weight == "latency":
|
|
52
|
+
total = _duration(span)
|
|
53
|
+
self_value = max(0, total - sum(_duration(k) for k in kids))
|
|
54
|
+
else:
|
|
55
|
+
own = _cost(span)
|
|
56
|
+
total = own + sum(f["value"] for f in child_frames)
|
|
57
|
+
self_value = own
|
|
58
|
+
return {
|
|
59
|
+
"name": span.name,
|
|
60
|
+
"kind": span.kind,
|
|
61
|
+
"span_id": span.span_id,
|
|
62
|
+
"status": span.status,
|
|
63
|
+
"value": total,
|
|
64
|
+
"self_value": self_value,
|
|
65
|
+
"children": child_frames,
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
roots = sorted(children.get(None, []), key=lambda s: s.start_ns)
|
|
69
|
+
root_frames = [build(r) for r in roots]
|
|
70
|
+
total = sum(f["value"] for f in root_frames)
|
|
71
|
+
return {
|
|
72
|
+
"name": "trace",
|
|
73
|
+
"kind": "trace",
|
|
74
|
+
"span_id": None,
|
|
75
|
+
"status": "ok",
|
|
76
|
+
"weight": weight,
|
|
77
|
+
"value": total,
|
|
78
|
+
"self_value": 0,
|
|
79
|
+
"children": root_frames,
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def waterfall(spans: list[Span]) -> list[dict[str, Any]]:
|
|
84
|
+
"""Flat timeline: one row per span with its depth, ordered by start."""
|
|
85
|
+
ids = {s.span_id for s in spans}
|
|
86
|
+
by_id = {s.span_id: s for s in spans}
|
|
87
|
+
|
|
88
|
+
def depth(span: Span) -> int:
|
|
89
|
+
level = 0
|
|
90
|
+
current = span
|
|
91
|
+
while current.parent_id in ids:
|
|
92
|
+
current = by_id[current.parent_id]
|
|
93
|
+
level += 1
|
|
94
|
+
if level > len(spans):
|
|
95
|
+
break
|
|
96
|
+
return level
|
|
97
|
+
|
|
98
|
+
rows = []
|
|
99
|
+
for span in sorted(spans, key=lambda s: s.start_ns):
|
|
100
|
+
rows.append(
|
|
101
|
+
{
|
|
102
|
+
"span_id": span.span_id,
|
|
103
|
+
"name": span.name,
|
|
104
|
+
"kind": span.kind,
|
|
105
|
+
"status": span.status,
|
|
106
|
+
"depth": depth(span),
|
|
107
|
+
"start_ns": span.start_ns,
|
|
108
|
+
"end_ns": span.end_ns,
|
|
109
|
+
"duration_ns": None if span.end_ns is None else _duration(span),
|
|
110
|
+
"cost_usd": span.attributes.get("cost_usd"),
|
|
111
|
+
}
|
|
112
|
+
)
|
|
113
|
+
return rows
|