tracemeter 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tracemeter/__init__.py ADDED
@@ -0,0 +1,41 @@
1
+ """TraceMeter: local-first, zero-infra cost & latency tracing for LLM pipelines.
2
+
3
+ Emits OpenTelemetry GenAI-compliant spans (`gen_ai.*` attributes) to a
4
+ local SQLite store -- no collector, no exporter config, no account
5
+ signup required to see a cost-annotated trace.
6
+
7
+ Quickstart:
8
+
9
+ import tracemeter
10
+
11
+ with tracemeter.span("my_pipeline"):
12
+ ...
13
+
14
+ @tracemeter.trace()
15
+ def my_step():
16
+ ...
17
+ """
18
+
19
+ from tracemeter.tracer import Tracer, get_default_tracer, trace
20
+ from tracemeter.pricing.engine import compute_cost
21
+ from tracemeter.integrations.openai_wrap import instrument_openai
22
+ from tracemeter.integrations.anthropic_wrap import instrument_anthropic
23
+ from tracemeter.integrations.litellm_wrap import instrument_litellm
24
+
25
+ __version__ = "0.1.0"
26
+
27
+ __all__ = [
28
+ "Tracer",
29
+ "get_default_tracer",
30
+ "trace",
31
+ "span",
32
+ "compute_cost",
33
+ "instrument_openai",
34
+ "instrument_anthropic",
35
+ "instrument_litellm",
36
+ ]
37
+
38
+
39
+ def span(name: str, **attributes):
40
+ """Shorthand for tracemeter.get_default_tracer().span(name, **attrs)."""
41
+ return get_default_tracer().span(name, **attributes)
tracemeter/cli.py ADDED
@@ -0,0 +1,68 @@
1
+ """tracemeter CLI: `tracemeter serve` launches the local dashboard.
2
+
3
+ Zero required config: run it, get a URL, see your traces. No collector,
4
+ no exporter, no account.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import argparse
10
+ import sys
11
+
12
+ from tracemeter.storage.sqlite_store import default_db_path
13
+
14
+
15
+ def _cmd_serve(args: argparse.Namespace) -> int:
16
+ try:
17
+ import uvicorn
18
+ except ImportError:
19
+ print(
20
+ "The dashboard server requires extra dependencies.\n"
21
+ "Install them with: pip install 'tracemeter[server]'",
22
+ file=sys.stderr,
23
+ )
24
+ return 1
25
+
26
+ from tracemeter.server.app import create_app
27
+ from tracemeter.storage.sqlite_store import SqliteStore
28
+
29
+ store = SqliteStore(args.db) if args.db else SqliteStore.default()
30
+ app = create_app(store=store)
31
+
32
+ print(f"TraceMeter dashboard: reading {store.db_path}")
33
+ print(f"Serving at http://{args.host}:{args.port}")
34
+ uvicorn.run(app, host=args.host, port=args.port, log_level="warning")
35
+ return 0
36
+
37
+
38
+ def _cmd_where(args: argparse.Namespace) -> int:
39
+ print(default_db_path())
40
+ return 0
41
+
42
+
43
+ def build_parser() -> argparse.ArgumentParser:
44
+ parser = argparse.ArgumentParser(prog="tracemeter")
45
+ subparsers = parser.add_subparsers(dest="command", required=True)
46
+
47
+ serve = subparsers.add_parser("serve", help="Launch the local dashboard")
48
+ serve.add_argument("--host", default="127.0.0.1")
49
+ serve.add_argument("--port", type=int, default=8765)
50
+ serve.add_argument(
51
+ "--db", default=None, help="Path to the SQLite trace DB (default: ~/.tracemeter/traces.db)"
52
+ )
53
+ serve.set_defaults(func=_cmd_serve)
54
+
55
+ where = subparsers.add_parser("where", help="Print the default trace DB path")
56
+ where.set_defaults(func=_cmd_where)
57
+
58
+ return parser
59
+
60
+
61
+ def main() -> None:
62
+ parser = build_parser()
63
+ args = parser.parse_args()
64
+ sys.exit(args.func(args))
65
+
66
+
67
+ if __name__ == "__main__":
68
+ main()
tracemeter/compare.py ADDED
@@ -0,0 +1,72 @@
1
+ """Run comparison: cost/latency diff between two traces (e.g. prompt v2
2
+ vs v1), matching steps by span name within each trace."""
3
+
4
+ from __future__ import annotations
5
+
6
+ from typing import Any
7
+
8
+ from tracemeter.storage.sqlite_store import SqliteStore
9
+
10
+
11
+ def _trace_totals(spans: list[dict[str, Any]]) -> dict[str, Any]:
12
+ cost = sum(s["attributes"].get("tracemeter.cost.usd") or 0.0 for s in spans)
13
+ if spans:
14
+ start = min(s["start_time"] for s in spans)
15
+ end = max(s["end_time"] or s["start_time"] for s in spans)
16
+ latency_ms = (end - start) * 1000.0
17
+ else:
18
+ latency_ms = 0.0
19
+ return {"cost_usd": cost, "latency_ms": latency_ms, "span_count": len(spans)}
20
+
21
+
22
+ def _steps_by_name(spans: list[dict[str, Any]]) -> dict[str, dict[str, Any]]:
23
+ """Groups non-root spans by name. The root span represents the whole
24
+ run and is already reflected in the totals, so including it here
25
+ would just add a noisy "only in A/B" row for every comparison."""
26
+ steps: dict[str, dict[str, Any]] = {}
27
+ for s in spans:
28
+ if s["parent_span_id"] is None:
29
+ continue
30
+ step = steps.setdefault(s["name"], {"cost_usd": 0.0, "latency_ms": 0.0, "count": 0})
31
+ step["cost_usd"] += s["attributes"].get("tracemeter.cost.usd") or 0.0
32
+ step["latency_ms"] += s["attributes"].get("tracemeter.latency_ms") or 0.0
33
+ step["count"] += 1
34
+ return steps
35
+
36
+
37
+ def compare_traces(store: SqliteStore, trace_id_a: str, trace_id_b: str) -> dict[str, Any]:
38
+ spans_a = store.get_trace_spans(trace_id_a)
39
+ spans_b = store.get_trace_spans(trace_id_b)
40
+
41
+ totals_a = _trace_totals(spans_a)
42
+ totals_b = _trace_totals(spans_b)
43
+
44
+ steps_a = _steps_by_name(spans_a)
45
+ steps_b = _steps_by_name(spans_b)
46
+
47
+ step_rows = []
48
+ for name in sorted(set(steps_a) | set(steps_b)):
49
+ a = steps_a.get(name, {"cost_usd": 0.0, "latency_ms": 0.0, "count": 0})
50
+ b = steps_b.get(name, {"cost_usd": 0.0, "latency_ms": 0.0, "count": 0})
51
+ step_rows.append(
52
+ {
53
+ "name": name,
54
+ "a_cost_usd": a["cost_usd"],
55
+ "b_cost_usd": b["cost_usd"],
56
+ "cost_delta_usd": b["cost_usd"] - a["cost_usd"],
57
+ "a_latency_ms": a["latency_ms"],
58
+ "b_latency_ms": b["latency_ms"],
59
+ "latency_delta_ms": b["latency_ms"] - a["latency_ms"],
60
+ "only_in": "b" if a["count"] == 0 else ("a" if b["count"] == 0 else None),
61
+ }
62
+ )
63
+
64
+ return {
65
+ "a": {"trace_id": trace_id_a, **totals_a},
66
+ "b": {"trace_id": trace_id_b, **totals_b},
67
+ "delta": {
68
+ "cost_usd": totals_b["cost_usd"] - totals_a["cost_usd"],
69
+ "latency_ms": totals_b["latency_ms"] - totals_a["latency_ms"],
70
+ },
71
+ "steps": step_rows,
72
+ }
@@ -0,0 +1 @@
1
+ """Auto-instrumentation for common LLM clients: openai, anthropic, litellm."""
@@ -0,0 +1,186 @@
1
+ """Shared plumbing for provider auto-instrumentation.
2
+
3
+ Wrapping strategy: patch the bound `create` method on a *client instance*
4
+ (not the class globally) so instrumenting one client never affects
5
+ another, and tests can instrument a fake client with no monkeypatching
6
+ of real SDK internals.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import inspect
12
+ import time
13
+ from typing import Any, Callable, Optional
14
+
15
+ from tracemeter import semconv
16
+ from tracemeter.pricing.engine import compute_cost
17
+ from tracemeter.tracer import Tracer, get_default_tracer
18
+
19
+
20
+ class _StreamSpanWrapper:
21
+ """Wraps a streaming response iterator so the span stays open across
22
+ iteration and closes (with usage/cost attrs, if available) once the
23
+ stream is exhausted or errors."""
24
+
25
+ def __init__(self, iterator: Any, span_ctx: Any, span: Any, on_chunk: Callable, on_done: Callable):
26
+ self._iterator = iterator
27
+ self._span_ctx = span_ctx
28
+ self._span = span
29
+ self._on_chunk = on_chunk
30
+ self._on_done = on_done
31
+ self._first_chunk_seen = False
32
+ self._closed = False
33
+
34
+ def __iter__(self):
35
+ return self
36
+
37
+ def __next__(self):
38
+ try:
39
+ chunk = next(self._iterator)
40
+ except StopIteration:
41
+ self._close()
42
+ raise
43
+ except BaseException as exc:
44
+ self._close(exc)
45
+ raise
46
+ if not self._first_chunk_seen:
47
+ self._first_chunk_seen = True
48
+ self._span.set_attribute(
49
+ semconv.TRACEMETER_TTFT_MS, (time.time() - self._span.start_time) * 1000.0
50
+ )
51
+ self._on_chunk(chunk)
52
+ return chunk
53
+
54
+ def _close(self, exc: Optional[BaseException] = None):
55
+ if self._closed:
56
+ return
57
+ self._closed = True
58
+ self._on_done()
59
+ self._span_ctx.__exit__(type(exc) if exc else None, exc, None)
60
+
61
+
62
+ def instrument_create_method(
63
+ client: Any,
64
+ attr_path: str,
65
+ system: str,
66
+ operation: str,
67
+ extract_model: Callable[[dict], Optional[str]],
68
+ extract_usage: Callable[[Any], tuple[int, int, int]],
69
+ is_streaming: Callable[[dict], bool],
70
+ on_stream_chunk_usage: Optional[Callable[[Any], Optional[tuple[int, int, int]]]] = None,
71
+ tracer: Optional[Tracer] = None,
72
+ ) -> None:
73
+ """Monkeypatch `client.<attr_path>` (e.g. "chat.completions.create") to
74
+ emit an OTel-GenAI-compliant span around every call.
75
+
76
+ extract_usage(response) -> (input_tokens, output_tokens, reasoning_tokens)
77
+ on_stream_chunk_usage(chunk) -> usage tuple or None; called per chunk to
78
+ find the final chunk's usage in streaming responses.
79
+ """
80
+ tracer = tracer or get_default_tracer()
81
+ parts = attr_path.split(".")
82
+ parent = client
83
+ for p in parts[:-1]:
84
+ parent = getattr(parent, p)
85
+ method_name = parts[-1]
86
+ original = getattr(parent, method_name)
87
+
88
+ if getattr(original, "_tracemeter_wrapped", False):
89
+ return # already instrumented
90
+
91
+ def _start_span(kwargs: dict):
92
+ model = extract_model(kwargs)
93
+ span_ctx = tracer.span(
94
+ f"{system}.{operation}",
95
+ **{
96
+ semconv.GEN_AI_SYSTEM: system,
97
+ semconv.GEN_AI_OPERATION_NAME: operation,
98
+ semconv.GEN_AI_REQUEST_MODEL: model,
99
+ },
100
+ )
101
+ span = span_ctx.__enter__()
102
+ if "max_tokens" in kwargs:
103
+ span.set_attribute(semconv.GEN_AI_REQUEST_MAX_TOKENS, kwargs["max_tokens"])
104
+ if "temperature" in kwargs:
105
+ span.set_attribute(semconv.GEN_AI_REQUEST_TEMPERATURE, kwargs["temperature"])
106
+ return span_ctx, span, model
107
+
108
+ def _finish_non_streaming(span_ctx, span, model, response):
109
+ in_tok, out_tok, reason_tok = extract_usage(response)
110
+ response_model = getattr(response, "model", None) or model
111
+ span.set_attributes(
112
+ {
113
+ semconv.GEN_AI_RESPONSE_MODEL: response_model,
114
+ semconv.GEN_AI_USAGE_INPUT_TOKENS: in_tok,
115
+ semconv.GEN_AI_USAGE_OUTPUT_TOKENS: out_tok,
116
+ }
117
+ )
118
+ if reason_tok:
119
+ span.set_attribute(semconv.GEN_AI_USAGE_REASONING_TOKENS, reason_tok)
120
+ _set_cost(span, system, response_model or model, in_tok, out_tok, reason_tok)
121
+ span_ctx.__exit__(None, None, None)
122
+
123
+ def _make_stream_wrapper(span_ctx, span, model, iterator):
124
+ usage_holder: dict[str, Any] = {}
125
+
126
+ def on_chunk(chunk):
127
+ if on_stream_chunk_usage:
128
+ usage = on_stream_chunk_usage(chunk)
129
+ if usage:
130
+ usage_holder["usage"] = usage
131
+
132
+ def on_done():
133
+ in_tok, out_tok, reason_tok = usage_holder.get("usage", (0, 0, 0))
134
+ span.set_attributes(
135
+ {
136
+ semconv.GEN_AI_RESPONSE_MODEL: model,
137
+ semconv.GEN_AI_USAGE_INPUT_TOKENS: in_tok,
138
+ semconv.GEN_AI_USAGE_OUTPUT_TOKENS: out_tok,
139
+ }
140
+ )
141
+ if reason_tok:
142
+ span.set_attribute(semconv.GEN_AI_USAGE_REASONING_TOKENS, reason_tok)
143
+ _set_cost(span, system, model, in_tok, out_tok, reason_tok)
144
+
145
+ return _StreamSpanWrapper(iterator, span_ctx, span, on_chunk, on_done)
146
+
147
+ if inspect.iscoroutinefunction(original):
148
+
149
+ async def wrapper(*args: Any, **kwargs: Any) -> Any:
150
+ span_ctx, span, model = _start_span(kwargs)
151
+ try:
152
+ response = await original(*args, **kwargs)
153
+ except BaseException as exc:
154
+ span_ctx.__exit__(type(exc), exc, None)
155
+ raise
156
+ if is_streaming(kwargs):
157
+ return _make_stream_wrapper(span_ctx, span, model, response)
158
+ _finish_non_streaming(span_ctx, span, model, response)
159
+ return response
160
+
161
+ else:
162
+
163
+ def wrapper(*args: Any, **kwargs: Any) -> Any:
164
+ span_ctx, span, model = _start_span(kwargs)
165
+ try:
166
+ response = original(*args, **kwargs)
167
+ except BaseException as exc:
168
+ span_ctx.__exit__(type(exc), exc, None)
169
+ raise
170
+ if is_streaming(kwargs):
171
+ return _make_stream_wrapper(span_ctx, span, model, response)
172
+ _finish_non_streaming(span_ctx, span, model, response)
173
+ return response
174
+
175
+ wrapper._tracemeter_wrapped = True # type: ignore[attr-defined]
176
+ setattr(parent, method_name, wrapper)
177
+
178
+
179
+ def _set_cost(span, system: str, model: Optional[str], in_tok: int, out_tok: int, reason_tok: int) -> None:
180
+ cost = compute_cost(
181
+ model, input_tokens=in_tok, output_tokens=out_tok, reasoning_tokens=reason_tok, system=system
182
+ )
183
+ if cost is None:
184
+ span.set_attribute(semconv.TRACEMETER_COST_UNKNOWN, True)
185
+ else:
186
+ span.set_attribute(semconv.TRACEMETER_COST_USD, cost)
@@ -0,0 +1,69 @@
1
+ """Auto-instrumentation for the `anthropic` client.
2
+
3
+ from anthropic import Anthropic
4
+ from tracemeter.integrations.anthropic_wrap import instrument_anthropic
5
+
6
+ client = Anthropic()
7
+ instrument_anthropic(client)
8
+
9
+ client.messages.create(...) # now traced automatically
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from typing import Any, Optional
15
+
16
+ from tracemeter.integrations._common import instrument_create_method
17
+ from tracemeter.tracer import Tracer
18
+
19
+
20
+ def _extract_model(kwargs: dict) -> Optional[str]:
21
+ return kwargs.get("model")
22
+
23
+
24
+ def _extract_usage(response: Any) -> tuple[int, int, int]:
25
+ usage = getattr(response, "usage", None)
26
+ if usage is None:
27
+ return (0, 0, 0)
28
+ input_tokens = getattr(usage, "input_tokens", 0) or 0
29
+ output_tokens = getattr(usage, "output_tokens", 0) or 0
30
+ return (input_tokens, output_tokens, 0)
31
+
32
+
33
+ def _is_streaming(kwargs: dict) -> bool:
34
+ return bool(kwargs.get("stream"))
35
+
36
+
37
+ def _on_stream_chunk_usage(chunk: Any) -> Optional[tuple[int, int, int]]:
38
+ # Anthropic streams a message_delta event carrying cumulative output
39
+ # usage near the end of the stream; message_start carries input usage.
40
+ usage = getattr(chunk, "usage", None)
41
+ if usage is None:
42
+ message = getattr(chunk, "message", None)
43
+ usage = getattr(message, "usage", None) if message else None
44
+ if usage is None:
45
+ return None
46
+ input_tokens = getattr(usage, "input_tokens", 0) or 0
47
+ output_tokens = getattr(usage, "output_tokens", 0) or 0
48
+ if not input_tokens and not output_tokens:
49
+ return None
50
+ return (input_tokens, output_tokens, 0)
51
+
52
+
53
+ def instrument_anthropic(client: Any, tracer: Optional[Tracer] = None) -> Any:
54
+ """Instrument an Anthropic (or AsyncAnthropic) client instance in place.
55
+
56
+ Wraps `client.messages.create`. Returns the same client for convenience.
57
+ """
58
+ instrument_create_method(
59
+ client,
60
+ "messages.create",
61
+ system="anthropic",
62
+ operation="chat",
63
+ extract_model=_extract_model,
64
+ extract_usage=_extract_usage,
65
+ is_streaming=_is_streaming,
66
+ on_stream_chunk_usage=_on_stream_chunk_usage,
67
+ tracer=tracer,
68
+ )
69
+ return client
@@ -0,0 +1,68 @@
1
+ """Auto-instrumentation for `litellm`.
2
+
3
+ Unlike the openai/anthropic SDKs, litellm's primary interface is a
4
+ module-level function (`litellm.completion`, `litellm.acompletion`)
5
+ rather than a client instance, so this patches the module directly
6
+ rather than a per-instance bound method.
7
+
8
+ import litellm
9
+ from tracemeter.integrations.litellm_wrap import instrument_litellm
10
+
11
+ instrument_litellm(litellm)
12
+
13
+ litellm.completion(model="gpt-4o-mini", messages=[...]) # traced
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from typing import Any, Optional
19
+
20
+ from tracemeter.integrations._common import instrument_create_method
21
+ from tracemeter.tracer import Tracer
22
+
23
+
24
+ def _extract_model(kwargs: dict) -> Optional[str]:
25
+ return kwargs.get("model")
26
+
27
+
28
+ def _extract_usage(response: Any) -> tuple[int, int, int]:
29
+ usage = getattr(response, "usage", None)
30
+ if usage is None and isinstance(response, dict):
31
+ usage = response.get("usage")
32
+ if usage is None:
33
+ return (0, 0, 0)
34
+ get = (lambda k: usage.get(k)) if isinstance(usage, dict) else (lambda k: getattr(usage, k, None))
35
+ return (get("prompt_tokens") or 0, get("completion_tokens") or 0, 0)
36
+
37
+
38
+ def _is_streaming(kwargs: dict) -> bool:
39
+ return bool(kwargs.get("stream"))
40
+
41
+
42
+ def instrument_litellm(litellm_module: Any, tracer: Optional[Tracer] = None) -> Any:
43
+ """Instrument the `litellm` module's `completion` (and `acompletion`,
44
+ if present) in place. litellm normalizes providers internally, so
45
+ `gen_ai.system` is reported as "litellm" -- the underlying provider is
46
+ still visible via the model string."""
47
+ instrument_create_method(
48
+ litellm_module,
49
+ "completion",
50
+ system="litellm",
51
+ operation="chat",
52
+ extract_model=_extract_model,
53
+ extract_usage=_extract_usage,
54
+ is_streaming=_is_streaming,
55
+ tracer=tracer,
56
+ )
57
+ if hasattr(litellm_module, "acompletion"):
58
+ instrument_create_method(
59
+ litellm_module,
60
+ "acompletion",
61
+ system="litellm",
62
+ operation="chat",
63
+ extract_model=_extract_model,
64
+ extract_usage=_extract_usage,
65
+ is_streaming=_is_streaming,
66
+ tracer=tracer,
67
+ )
68
+ return litellm_module
@@ -0,0 +1,79 @@
1
+ """Auto-instrumentation for the `openai` client (>=1.0 SDK).
2
+
3
+ from openai import OpenAI
4
+ from tracemeter.integrations.openai_wrap import instrument_openai
5
+
6
+ client = OpenAI()
7
+ instrument_openai(client)
8
+
9
+ client.chat.completions.create(...) # now traced automatically
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from typing import Any, Optional
15
+
16
+ from tracemeter.integrations._common import instrument_create_method
17
+ from tracemeter.tracer import Tracer
18
+
19
+
20
+ def _extract_model(kwargs: dict) -> Optional[str]:
21
+ return kwargs.get("model")
22
+
23
+
24
+ def _extract_usage(response: Any) -> tuple[int, int, int]:
25
+ usage = getattr(response, "usage", None)
26
+ if usage is None:
27
+ return (0, 0, 0)
28
+ input_tokens = getattr(usage, "prompt_tokens", 0) or 0
29
+ output_tokens = getattr(usage, "completion_tokens", 0) or 0
30
+ reasoning_tokens = 0
31
+ details = getattr(usage, "completion_tokens_details", None)
32
+ if details is not None:
33
+ reasoning_tokens = getattr(details, "reasoning_tokens", 0) or 0
34
+ return (input_tokens, output_tokens, reasoning_tokens)
35
+
36
+
37
+ def _is_streaming(kwargs: dict) -> bool:
38
+ return bool(kwargs.get("stream"))
39
+
40
+
41
+ def _on_stream_chunk_usage(chunk: Any) -> Optional[tuple[int, int, int]]:
42
+ # Only populated on the final chunk when `stream_options={"include_usage": True}`.
43
+ usage = getattr(chunk, "usage", None)
44
+ if usage is None:
45
+ return None
46
+ input_tokens = getattr(usage, "prompt_tokens", 0) or 0
47
+ output_tokens = getattr(usage, "completion_tokens", 0) or 0
48
+ return (input_tokens, output_tokens, 0)
49
+
50
+
51
+ def instrument_openai(client: Any, tracer: Optional[Tracer] = None) -> Any:
52
+ """Instrument an OpenAI (or AsyncOpenAI) client instance in place.
53
+
54
+ Wraps `client.chat.completions.create` and `client.embeddings.create`.
55
+ Returns the same client for convenience.
56
+ """
57
+ instrument_create_method(
58
+ client,
59
+ "chat.completions.create",
60
+ system="openai",
61
+ operation="chat",
62
+ extract_model=_extract_model,
63
+ extract_usage=_extract_usage,
64
+ is_streaming=_is_streaming,
65
+ on_stream_chunk_usage=_on_stream_chunk_usage,
66
+ tracer=tracer,
67
+ )
68
+ if hasattr(client, "embeddings"):
69
+ instrument_create_method(
70
+ client,
71
+ "embeddings.create",
72
+ system="openai",
73
+ operation="embeddings",
74
+ extract_model=_extract_model,
75
+ extract_usage=_extract_usage,
76
+ is_streaming=lambda kwargs: False,
77
+ tracer=tracer,
78
+ )
79
+ return client
@@ -0,0 +1,3 @@
1
+ from tracemeter.pricing.engine import PricingEngine, compute_cost
2
+
3
+ __all__ = ["PricingEngine", "compute_cost"]