tracemeter 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tracemeter/__init__.py +41 -0
- tracemeter/cli.py +68 -0
- tracemeter/compare.py +72 -0
- tracemeter/integrations/__init__.py +1 -0
- tracemeter/integrations/_common.py +186 -0
- tracemeter/integrations/anthropic_wrap.py +69 -0
- tracemeter/integrations/litellm_wrap.py +68 -0
- tracemeter/integrations/openai_wrap.py +79 -0
- tracemeter/pricing/__init__.py +3 -0
- tracemeter/pricing/engine.py +116 -0
- tracemeter/pricing/prices.json +45 -0
- tracemeter/semconv.py +48 -0
- tracemeter/server/__init__.py +0 -0
- tracemeter/server/app.py +146 -0
- tracemeter/server/otlp.py +270 -0
- tracemeter/server/static/index.html +421 -0
- tracemeter/storage/__init__.py +3 -0
- tracemeter/storage/sqlite_store.py +244 -0
- tracemeter/tracer.py +160 -0
- tracemeter-0.1.0.dist-info/METADATA +140 -0
- tracemeter-0.1.0.dist-info/RECORD +24 -0
- tracemeter-0.1.0.dist-info/WHEEL +4 -0
- tracemeter-0.1.0.dist-info/entry_points.txt +2 -0
- tracemeter-0.1.0.dist-info/licenses/LICENSE +202 -0
tracemeter/__init__.py
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""TraceMeter: local-first, zero-infra cost & latency tracing for LLM pipelines.
|
|
2
|
+
|
|
3
|
+
Emits OpenTelemetry GenAI-compliant spans (`gen_ai.*` attributes) to a
|
|
4
|
+
local SQLite store -- no collector, no exporter config, no account
|
|
5
|
+
signup required to see a cost-annotated trace.
|
|
6
|
+
|
|
7
|
+
Quickstart:
|
|
8
|
+
|
|
9
|
+
import tracemeter
|
|
10
|
+
|
|
11
|
+
with tracemeter.span("my_pipeline"):
|
|
12
|
+
...
|
|
13
|
+
|
|
14
|
+
@tracemeter.trace()
|
|
15
|
+
def my_step():
|
|
16
|
+
...
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from tracemeter.tracer import Tracer, get_default_tracer, trace
|
|
20
|
+
from tracemeter.pricing.engine import compute_cost
|
|
21
|
+
from tracemeter.integrations.openai_wrap import instrument_openai
|
|
22
|
+
from tracemeter.integrations.anthropic_wrap import instrument_anthropic
|
|
23
|
+
from tracemeter.integrations.litellm_wrap import instrument_litellm
|
|
24
|
+
|
|
25
|
+
__version__ = "0.1.0"
|
|
26
|
+
|
|
27
|
+
__all__ = [
|
|
28
|
+
"Tracer",
|
|
29
|
+
"get_default_tracer",
|
|
30
|
+
"trace",
|
|
31
|
+
"span",
|
|
32
|
+
"compute_cost",
|
|
33
|
+
"instrument_openai",
|
|
34
|
+
"instrument_anthropic",
|
|
35
|
+
"instrument_litellm",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def span(name: str, **attributes):
|
|
40
|
+
"""Shorthand for tracemeter.get_default_tracer().span(name, **attrs)."""
|
|
41
|
+
return get_default_tracer().span(name, **attributes)
|
tracemeter/cli.py
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""tracemeter CLI: `tracemeter serve` launches the local dashboard.
|
|
2
|
+
|
|
3
|
+
Zero required config: run it, get a URL, see your traces. No collector,
|
|
4
|
+
no exporter, no account.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import argparse
|
|
10
|
+
import sys
|
|
11
|
+
|
|
12
|
+
from tracemeter.storage.sqlite_store import default_db_path
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _cmd_serve(args: argparse.Namespace) -> int:
|
|
16
|
+
try:
|
|
17
|
+
import uvicorn
|
|
18
|
+
except ImportError:
|
|
19
|
+
print(
|
|
20
|
+
"The dashboard server requires extra dependencies.\n"
|
|
21
|
+
"Install them with: pip install 'tracemeter[server]'",
|
|
22
|
+
file=sys.stderr,
|
|
23
|
+
)
|
|
24
|
+
return 1
|
|
25
|
+
|
|
26
|
+
from tracemeter.server.app import create_app
|
|
27
|
+
from tracemeter.storage.sqlite_store import SqliteStore
|
|
28
|
+
|
|
29
|
+
store = SqliteStore(args.db) if args.db else SqliteStore.default()
|
|
30
|
+
app = create_app(store=store)
|
|
31
|
+
|
|
32
|
+
print(f"TraceMeter dashboard: reading {store.db_path}")
|
|
33
|
+
print(f"Serving at http://{args.host}:{args.port}")
|
|
34
|
+
uvicorn.run(app, host=args.host, port=args.port, log_level="warning")
|
|
35
|
+
return 0
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _cmd_where(args: argparse.Namespace) -> int:
|
|
39
|
+
print(default_db_path())
|
|
40
|
+
return 0
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
44
|
+
parser = argparse.ArgumentParser(prog="tracemeter")
|
|
45
|
+
subparsers = parser.add_subparsers(dest="command", required=True)
|
|
46
|
+
|
|
47
|
+
serve = subparsers.add_parser("serve", help="Launch the local dashboard")
|
|
48
|
+
serve.add_argument("--host", default="127.0.0.1")
|
|
49
|
+
serve.add_argument("--port", type=int, default=8765)
|
|
50
|
+
serve.add_argument(
|
|
51
|
+
"--db", default=None, help="Path to the SQLite trace DB (default: ~/.tracemeter/traces.db)"
|
|
52
|
+
)
|
|
53
|
+
serve.set_defaults(func=_cmd_serve)
|
|
54
|
+
|
|
55
|
+
where = subparsers.add_parser("where", help="Print the default trace DB path")
|
|
56
|
+
where.set_defaults(func=_cmd_where)
|
|
57
|
+
|
|
58
|
+
return parser
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def main() -> None:
|
|
62
|
+
parser = build_parser()
|
|
63
|
+
args = parser.parse_args()
|
|
64
|
+
sys.exit(args.func(args))
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
if __name__ == "__main__":
|
|
68
|
+
main()
|
tracemeter/compare.py
ADDED
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"""Run comparison: cost/latency diff between two traces (e.g. prompt v2
|
|
2
|
+
vs v1), matching steps by span name within each trace."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from tracemeter.storage.sqlite_store import SqliteStore
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _trace_totals(spans: list[dict[str, Any]]) -> dict[str, Any]:
|
|
12
|
+
cost = sum(s["attributes"].get("tracemeter.cost.usd") or 0.0 for s in spans)
|
|
13
|
+
if spans:
|
|
14
|
+
start = min(s["start_time"] for s in spans)
|
|
15
|
+
end = max(s["end_time"] or s["start_time"] for s in spans)
|
|
16
|
+
latency_ms = (end - start) * 1000.0
|
|
17
|
+
else:
|
|
18
|
+
latency_ms = 0.0
|
|
19
|
+
return {"cost_usd": cost, "latency_ms": latency_ms, "span_count": len(spans)}
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _steps_by_name(spans: list[dict[str, Any]]) -> dict[str, dict[str, Any]]:
|
|
23
|
+
"""Groups non-root spans by name. The root span represents the whole
|
|
24
|
+
run and is already reflected in the totals, so including it here
|
|
25
|
+
would just add a noisy "only in A/B" row for every comparison."""
|
|
26
|
+
steps: dict[str, dict[str, Any]] = {}
|
|
27
|
+
for s in spans:
|
|
28
|
+
if s["parent_span_id"] is None:
|
|
29
|
+
continue
|
|
30
|
+
step = steps.setdefault(s["name"], {"cost_usd": 0.0, "latency_ms": 0.0, "count": 0})
|
|
31
|
+
step["cost_usd"] += s["attributes"].get("tracemeter.cost.usd") or 0.0
|
|
32
|
+
step["latency_ms"] += s["attributes"].get("tracemeter.latency_ms") or 0.0
|
|
33
|
+
step["count"] += 1
|
|
34
|
+
return steps
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def compare_traces(store: SqliteStore, trace_id_a: str, trace_id_b: str) -> dict[str, Any]:
|
|
38
|
+
spans_a = store.get_trace_spans(trace_id_a)
|
|
39
|
+
spans_b = store.get_trace_spans(trace_id_b)
|
|
40
|
+
|
|
41
|
+
totals_a = _trace_totals(spans_a)
|
|
42
|
+
totals_b = _trace_totals(spans_b)
|
|
43
|
+
|
|
44
|
+
steps_a = _steps_by_name(spans_a)
|
|
45
|
+
steps_b = _steps_by_name(spans_b)
|
|
46
|
+
|
|
47
|
+
step_rows = []
|
|
48
|
+
for name in sorted(set(steps_a) | set(steps_b)):
|
|
49
|
+
a = steps_a.get(name, {"cost_usd": 0.0, "latency_ms": 0.0, "count": 0})
|
|
50
|
+
b = steps_b.get(name, {"cost_usd": 0.0, "latency_ms": 0.0, "count": 0})
|
|
51
|
+
step_rows.append(
|
|
52
|
+
{
|
|
53
|
+
"name": name,
|
|
54
|
+
"a_cost_usd": a["cost_usd"],
|
|
55
|
+
"b_cost_usd": b["cost_usd"],
|
|
56
|
+
"cost_delta_usd": b["cost_usd"] - a["cost_usd"],
|
|
57
|
+
"a_latency_ms": a["latency_ms"],
|
|
58
|
+
"b_latency_ms": b["latency_ms"],
|
|
59
|
+
"latency_delta_ms": b["latency_ms"] - a["latency_ms"],
|
|
60
|
+
"only_in": "b" if a["count"] == 0 else ("a" if b["count"] == 0 else None),
|
|
61
|
+
}
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
return {
|
|
65
|
+
"a": {"trace_id": trace_id_a, **totals_a},
|
|
66
|
+
"b": {"trace_id": trace_id_b, **totals_b},
|
|
67
|
+
"delta": {
|
|
68
|
+
"cost_usd": totals_b["cost_usd"] - totals_a["cost_usd"],
|
|
69
|
+
"latency_ms": totals_b["latency_ms"] - totals_a["latency_ms"],
|
|
70
|
+
},
|
|
71
|
+
"steps": step_rows,
|
|
72
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Auto-instrumentation for common LLM clients: openai, anthropic, litellm."""
|
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
"""Shared plumbing for provider auto-instrumentation.
|
|
2
|
+
|
|
3
|
+
Wrapping strategy: patch the bound `create` method on a *client instance*
|
|
4
|
+
(not the class globally) so instrumenting one client never affects
|
|
5
|
+
another, and tests can instrument a fake client with no monkeypatching
|
|
6
|
+
of real SDK internals.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import inspect
|
|
12
|
+
import time
|
|
13
|
+
from typing import Any, Callable, Optional
|
|
14
|
+
|
|
15
|
+
from tracemeter import semconv
|
|
16
|
+
from tracemeter.pricing.engine import compute_cost
|
|
17
|
+
from tracemeter.tracer import Tracer, get_default_tracer
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class _StreamSpanWrapper:
|
|
21
|
+
"""Wraps a streaming response iterator so the span stays open across
|
|
22
|
+
iteration and closes (with usage/cost attrs, if available) once the
|
|
23
|
+
stream is exhausted or errors."""
|
|
24
|
+
|
|
25
|
+
def __init__(self, iterator: Any, span_ctx: Any, span: Any, on_chunk: Callable, on_done: Callable):
|
|
26
|
+
self._iterator = iterator
|
|
27
|
+
self._span_ctx = span_ctx
|
|
28
|
+
self._span = span
|
|
29
|
+
self._on_chunk = on_chunk
|
|
30
|
+
self._on_done = on_done
|
|
31
|
+
self._first_chunk_seen = False
|
|
32
|
+
self._closed = False
|
|
33
|
+
|
|
34
|
+
def __iter__(self):
|
|
35
|
+
return self
|
|
36
|
+
|
|
37
|
+
def __next__(self):
|
|
38
|
+
try:
|
|
39
|
+
chunk = next(self._iterator)
|
|
40
|
+
except StopIteration:
|
|
41
|
+
self._close()
|
|
42
|
+
raise
|
|
43
|
+
except BaseException as exc:
|
|
44
|
+
self._close(exc)
|
|
45
|
+
raise
|
|
46
|
+
if not self._first_chunk_seen:
|
|
47
|
+
self._first_chunk_seen = True
|
|
48
|
+
self._span.set_attribute(
|
|
49
|
+
semconv.TRACEMETER_TTFT_MS, (time.time() - self._span.start_time) * 1000.0
|
|
50
|
+
)
|
|
51
|
+
self._on_chunk(chunk)
|
|
52
|
+
return chunk
|
|
53
|
+
|
|
54
|
+
def _close(self, exc: Optional[BaseException] = None):
|
|
55
|
+
if self._closed:
|
|
56
|
+
return
|
|
57
|
+
self._closed = True
|
|
58
|
+
self._on_done()
|
|
59
|
+
self._span_ctx.__exit__(type(exc) if exc else None, exc, None)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def instrument_create_method(
|
|
63
|
+
client: Any,
|
|
64
|
+
attr_path: str,
|
|
65
|
+
system: str,
|
|
66
|
+
operation: str,
|
|
67
|
+
extract_model: Callable[[dict], Optional[str]],
|
|
68
|
+
extract_usage: Callable[[Any], tuple[int, int, int]],
|
|
69
|
+
is_streaming: Callable[[dict], bool],
|
|
70
|
+
on_stream_chunk_usage: Optional[Callable[[Any], Optional[tuple[int, int, int]]]] = None,
|
|
71
|
+
tracer: Optional[Tracer] = None,
|
|
72
|
+
) -> None:
|
|
73
|
+
"""Monkeypatch `client.<attr_path>` (e.g. "chat.completions.create") to
|
|
74
|
+
emit an OTel-GenAI-compliant span around every call.
|
|
75
|
+
|
|
76
|
+
extract_usage(response) -> (input_tokens, output_tokens, reasoning_tokens)
|
|
77
|
+
on_stream_chunk_usage(chunk) -> usage tuple or None; called per chunk to
|
|
78
|
+
find the final chunk's usage in streaming responses.
|
|
79
|
+
"""
|
|
80
|
+
tracer = tracer or get_default_tracer()
|
|
81
|
+
parts = attr_path.split(".")
|
|
82
|
+
parent = client
|
|
83
|
+
for p in parts[:-1]:
|
|
84
|
+
parent = getattr(parent, p)
|
|
85
|
+
method_name = parts[-1]
|
|
86
|
+
original = getattr(parent, method_name)
|
|
87
|
+
|
|
88
|
+
if getattr(original, "_tracemeter_wrapped", False):
|
|
89
|
+
return # already instrumented
|
|
90
|
+
|
|
91
|
+
def _start_span(kwargs: dict):
|
|
92
|
+
model = extract_model(kwargs)
|
|
93
|
+
span_ctx = tracer.span(
|
|
94
|
+
f"{system}.{operation}",
|
|
95
|
+
**{
|
|
96
|
+
semconv.GEN_AI_SYSTEM: system,
|
|
97
|
+
semconv.GEN_AI_OPERATION_NAME: operation,
|
|
98
|
+
semconv.GEN_AI_REQUEST_MODEL: model,
|
|
99
|
+
},
|
|
100
|
+
)
|
|
101
|
+
span = span_ctx.__enter__()
|
|
102
|
+
if "max_tokens" in kwargs:
|
|
103
|
+
span.set_attribute(semconv.GEN_AI_REQUEST_MAX_TOKENS, kwargs["max_tokens"])
|
|
104
|
+
if "temperature" in kwargs:
|
|
105
|
+
span.set_attribute(semconv.GEN_AI_REQUEST_TEMPERATURE, kwargs["temperature"])
|
|
106
|
+
return span_ctx, span, model
|
|
107
|
+
|
|
108
|
+
def _finish_non_streaming(span_ctx, span, model, response):
|
|
109
|
+
in_tok, out_tok, reason_tok = extract_usage(response)
|
|
110
|
+
response_model = getattr(response, "model", None) or model
|
|
111
|
+
span.set_attributes(
|
|
112
|
+
{
|
|
113
|
+
semconv.GEN_AI_RESPONSE_MODEL: response_model,
|
|
114
|
+
semconv.GEN_AI_USAGE_INPUT_TOKENS: in_tok,
|
|
115
|
+
semconv.GEN_AI_USAGE_OUTPUT_TOKENS: out_tok,
|
|
116
|
+
}
|
|
117
|
+
)
|
|
118
|
+
if reason_tok:
|
|
119
|
+
span.set_attribute(semconv.GEN_AI_USAGE_REASONING_TOKENS, reason_tok)
|
|
120
|
+
_set_cost(span, system, response_model or model, in_tok, out_tok, reason_tok)
|
|
121
|
+
span_ctx.__exit__(None, None, None)
|
|
122
|
+
|
|
123
|
+
def _make_stream_wrapper(span_ctx, span, model, iterator):
|
|
124
|
+
usage_holder: dict[str, Any] = {}
|
|
125
|
+
|
|
126
|
+
def on_chunk(chunk):
|
|
127
|
+
if on_stream_chunk_usage:
|
|
128
|
+
usage = on_stream_chunk_usage(chunk)
|
|
129
|
+
if usage:
|
|
130
|
+
usage_holder["usage"] = usage
|
|
131
|
+
|
|
132
|
+
def on_done():
|
|
133
|
+
in_tok, out_tok, reason_tok = usage_holder.get("usage", (0, 0, 0))
|
|
134
|
+
span.set_attributes(
|
|
135
|
+
{
|
|
136
|
+
semconv.GEN_AI_RESPONSE_MODEL: model,
|
|
137
|
+
semconv.GEN_AI_USAGE_INPUT_TOKENS: in_tok,
|
|
138
|
+
semconv.GEN_AI_USAGE_OUTPUT_TOKENS: out_tok,
|
|
139
|
+
}
|
|
140
|
+
)
|
|
141
|
+
if reason_tok:
|
|
142
|
+
span.set_attribute(semconv.GEN_AI_USAGE_REASONING_TOKENS, reason_tok)
|
|
143
|
+
_set_cost(span, system, model, in_tok, out_tok, reason_tok)
|
|
144
|
+
|
|
145
|
+
return _StreamSpanWrapper(iterator, span_ctx, span, on_chunk, on_done)
|
|
146
|
+
|
|
147
|
+
if inspect.iscoroutinefunction(original):
|
|
148
|
+
|
|
149
|
+
async def wrapper(*args: Any, **kwargs: Any) -> Any:
|
|
150
|
+
span_ctx, span, model = _start_span(kwargs)
|
|
151
|
+
try:
|
|
152
|
+
response = await original(*args, **kwargs)
|
|
153
|
+
except BaseException as exc:
|
|
154
|
+
span_ctx.__exit__(type(exc), exc, None)
|
|
155
|
+
raise
|
|
156
|
+
if is_streaming(kwargs):
|
|
157
|
+
return _make_stream_wrapper(span_ctx, span, model, response)
|
|
158
|
+
_finish_non_streaming(span_ctx, span, model, response)
|
|
159
|
+
return response
|
|
160
|
+
|
|
161
|
+
else:
|
|
162
|
+
|
|
163
|
+
def wrapper(*args: Any, **kwargs: Any) -> Any:
|
|
164
|
+
span_ctx, span, model = _start_span(kwargs)
|
|
165
|
+
try:
|
|
166
|
+
response = original(*args, **kwargs)
|
|
167
|
+
except BaseException as exc:
|
|
168
|
+
span_ctx.__exit__(type(exc), exc, None)
|
|
169
|
+
raise
|
|
170
|
+
if is_streaming(kwargs):
|
|
171
|
+
return _make_stream_wrapper(span_ctx, span, model, response)
|
|
172
|
+
_finish_non_streaming(span_ctx, span, model, response)
|
|
173
|
+
return response
|
|
174
|
+
|
|
175
|
+
wrapper._tracemeter_wrapped = True # type: ignore[attr-defined]
|
|
176
|
+
setattr(parent, method_name, wrapper)
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _set_cost(span, system: str, model: Optional[str], in_tok: int, out_tok: int, reason_tok: int) -> None:
|
|
180
|
+
cost = compute_cost(
|
|
181
|
+
model, input_tokens=in_tok, output_tokens=out_tok, reasoning_tokens=reason_tok, system=system
|
|
182
|
+
)
|
|
183
|
+
if cost is None:
|
|
184
|
+
span.set_attribute(semconv.TRACEMETER_COST_UNKNOWN, True)
|
|
185
|
+
else:
|
|
186
|
+
span.set_attribute(semconv.TRACEMETER_COST_USD, cost)
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Auto-instrumentation for the `anthropic` client.
|
|
2
|
+
|
|
3
|
+
from anthropic import Anthropic
|
|
4
|
+
from tracemeter.integrations.anthropic_wrap import instrument_anthropic
|
|
5
|
+
|
|
6
|
+
client = Anthropic()
|
|
7
|
+
instrument_anthropic(client)
|
|
8
|
+
|
|
9
|
+
client.messages.create(...) # now traced automatically
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from typing import Any, Optional
|
|
15
|
+
|
|
16
|
+
from tracemeter.integrations._common import instrument_create_method
|
|
17
|
+
from tracemeter.tracer import Tracer
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _extract_model(kwargs: dict) -> Optional[str]:
|
|
21
|
+
return kwargs.get("model")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _extract_usage(response: Any) -> tuple[int, int, int]:
|
|
25
|
+
usage = getattr(response, "usage", None)
|
|
26
|
+
if usage is None:
|
|
27
|
+
return (0, 0, 0)
|
|
28
|
+
input_tokens = getattr(usage, "input_tokens", 0) or 0
|
|
29
|
+
output_tokens = getattr(usage, "output_tokens", 0) or 0
|
|
30
|
+
return (input_tokens, output_tokens, 0)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _is_streaming(kwargs: dict) -> bool:
|
|
34
|
+
return bool(kwargs.get("stream"))
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _on_stream_chunk_usage(chunk: Any) -> Optional[tuple[int, int, int]]:
|
|
38
|
+
# Anthropic streams a message_delta event carrying cumulative output
|
|
39
|
+
# usage near the end of the stream; message_start carries input usage.
|
|
40
|
+
usage = getattr(chunk, "usage", None)
|
|
41
|
+
if usage is None:
|
|
42
|
+
message = getattr(chunk, "message", None)
|
|
43
|
+
usage = getattr(message, "usage", None) if message else None
|
|
44
|
+
if usage is None:
|
|
45
|
+
return None
|
|
46
|
+
input_tokens = getattr(usage, "input_tokens", 0) or 0
|
|
47
|
+
output_tokens = getattr(usage, "output_tokens", 0) or 0
|
|
48
|
+
if not input_tokens and not output_tokens:
|
|
49
|
+
return None
|
|
50
|
+
return (input_tokens, output_tokens, 0)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def instrument_anthropic(client: Any, tracer: Optional[Tracer] = None) -> Any:
|
|
54
|
+
"""Instrument an Anthropic (or AsyncAnthropic) client instance in place.
|
|
55
|
+
|
|
56
|
+
Wraps `client.messages.create`. Returns the same client for convenience.
|
|
57
|
+
"""
|
|
58
|
+
instrument_create_method(
|
|
59
|
+
client,
|
|
60
|
+
"messages.create",
|
|
61
|
+
system="anthropic",
|
|
62
|
+
operation="chat",
|
|
63
|
+
extract_model=_extract_model,
|
|
64
|
+
extract_usage=_extract_usage,
|
|
65
|
+
is_streaming=_is_streaming,
|
|
66
|
+
on_stream_chunk_usage=_on_stream_chunk_usage,
|
|
67
|
+
tracer=tracer,
|
|
68
|
+
)
|
|
69
|
+
return client
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""Auto-instrumentation for `litellm`.
|
|
2
|
+
|
|
3
|
+
Unlike the openai/anthropic SDKs, litellm's primary interface is a
|
|
4
|
+
module-level function (`litellm.completion`, `litellm.acompletion`)
|
|
5
|
+
rather than a client instance, so this patches the module directly
|
|
6
|
+
rather than a per-instance bound method.
|
|
7
|
+
|
|
8
|
+
import litellm
|
|
9
|
+
from tracemeter.integrations.litellm_wrap import instrument_litellm
|
|
10
|
+
|
|
11
|
+
instrument_litellm(litellm)
|
|
12
|
+
|
|
13
|
+
litellm.completion(model="gpt-4o-mini", messages=[...]) # traced
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from typing import Any, Optional
|
|
19
|
+
|
|
20
|
+
from tracemeter.integrations._common import instrument_create_method
|
|
21
|
+
from tracemeter.tracer import Tracer
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _extract_model(kwargs: dict) -> Optional[str]:
|
|
25
|
+
return kwargs.get("model")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _extract_usage(response: Any) -> tuple[int, int, int]:
|
|
29
|
+
usage = getattr(response, "usage", None)
|
|
30
|
+
if usage is None and isinstance(response, dict):
|
|
31
|
+
usage = response.get("usage")
|
|
32
|
+
if usage is None:
|
|
33
|
+
return (0, 0, 0)
|
|
34
|
+
get = (lambda k: usage.get(k)) if isinstance(usage, dict) else (lambda k: getattr(usage, k, None))
|
|
35
|
+
return (get("prompt_tokens") or 0, get("completion_tokens") or 0, 0)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _is_streaming(kwargs: dict) -> bool:
|
|
39
|
+
return bool(kwargs.get("stream"))
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def instrument_litellm(litellm_module: Any, tracer: Optional[Tracer] = None) -> Any:
|
|
43
|
+
"""Instrument the `litellm` module's `completion` (and `acompletion`,
|
|
44
|
+
if present) in place. litellm normalizes providers internally, so
|
|
45
|
+
`gen_ai.system` is reported as "litellm" -- the underlying provider is
|
|
46
|
+
still visible via the model string."""
|
|
47
|
+
instrument_create_method(
|
|
48
|
+
litellm_module,
|
|
49
|
+
"completion",
|
|
50
|
+
system="litellm",
|
|
51
|
+
operation="chat",
|
|
52
|
+
extract_model=_extract_model,
|
|
53
|
+
extract_usage=_extract_usage,
|
|
54
|
+
is_streaming=_is_streaming,
|
|
55
|
+
tracer=tracer,
|
|
56
|
+
)
|
|
57
|
+
if hasattr(litellm_module, "acompletion"):
|
|
58
|
+
instrument_create_method(
|
|
59
|
+
litellm_module,
|
|
60
|
+
"acompletion",
|
|
61
|
+
system="litellm",
|
|
62
|
+
operation="chat",
|
|
63
|
+
extract_model=_extract_model,
|
|
64
|
+
extract_usage=_extract_usage,
|
|
65
|
+
is_streaming=_is_streaming,
|
|
66
|
+
tracer=tracer,
|
|
67
|
+
)
|
|
68
|
+
return litellm_module
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
"""Auto-instrumentation for the `openai` client (>=1.0 SDK).
|
|
2
|
+
|
|
3
|
+
from openai import OpenAI
|
|
4
|
+
from tracemeter.integrations.openai_wrap import instrument_openai
|
|
5
|
+
|
|
6
|
+
client = OpenAI()
|
|
7
|
+
instrument_openai(client)
|
|
8
|
+
|
|
9
|
+
client.chat.completions.create(...) # now traced automatically
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from typing import Any, Optional
|
|
15
|
+
|
|
16
|
+
from tracemeter.integrations._common import instrument_create_method
|
|
17
|
+
from tracemeter.tracer import Tracer
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _extract_model(kwargs: dict) -> Optional[str]:
|
|
21
|
+
return kwargs.get("model")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _extract_usage(response: Any) -> tuple[int, int, int]:
|
|
25
|
+
usage = getattr(response, "usage", None)
|
|
26
|
+
if usage is None:
|
|
27
|
+
return (0, 0, 0)
|
|
28
|
+
input_tokens = getattr(usage, "prompt_tokens", 0) or 0
|
|
29
|
+
output_tokens = getattr(usage, "completion_tokens", 0) or 0
|
|
30
|
+
reasoning_tokens = 0
|
|
31
|
+
details = getattr(usage, "completion_tokens_details", None)
|
|
32
|
+
if details is not None:
|
|
33
|
+
reasoning_tokens = getattr(details, "reasoning_tokens", 0) or 0
|
|
34
|
+
return (input_tokens, output_tokens, reasoning_tokens)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _is_streaming(kwargs: dict) -> bool:
|
|
38
|
+
return bool(kwargs.get("stream"))
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _on_stream_chunk_usage(chunk: Any) -> Optional[tuple[int, int, int]]:
|
|
42
|
+
# Only populated on the final chunk when `stream_options={"include_usage": True}`.
|
|
43
|
+
usage = getattr(chunk, "usage", None)
|
|
44
|
+
if usage is None:
|
|
45
|
+
return None
|
|
46
|
+
input_tokens = getattr(usage, "prompt_tokens", 0) or 0
|
|
47
|
+
output_tokens = getattr(usage, "completion_tokens", 0) or 0
|
|
48
|
+
return (input_tokens, output_tokens, 0)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def instrument_openai(client: Any, tracer: Optional[Tracer] = None) -> Any:
|
|
52
|
+
"""Instrument an OpenAI (or AsyncOpenAI) client instance in place.
|
|
53
|
+
|
|
54
|
+
Wraps `client.chat.completions.create` and `client.embeddings.create`.
|
|
55
|
+
Returns the same client for convenience.
|
|
56
|
+
"""
|
|
57
|
+
instrument_create_method(
|
|
58
|
+
client,
|
|
59
|
+
"chat.completions.create",
|
|
60
|
+
system="openai",
|
|
61
|
+
operation="chat",
|
|
62
|
+
extract_model=_extract_model,
|
|
63
|
+
extract_usage=_extract_usage,
|
|
64
|
+
is_streaming=_is_streaming,
|
|
65
|
+
on_stream_chunk_usage=_on_stream_chunk_usage,
|
|
66
|
+
tracer=tracer,
|
|
67
|
+
)
|
|
68
|
+
if hasattr(client, "embeddings"):
|
|
69
|
+
instrument_create_method(
|
|
70
|
+
client,
|
|
71
|
+
"embeddings.create",
|
|
72
|
+
system="openai",
|
|
73
|
+
operation="embeddings",
|
|
74
|
+
extract_model=_extract_model,
|
|
75
|
+
extract_usage=_extract_usage,
|
|
76
|
+
is_streaming=lambda kwargs: False,
|
|
77
|
+
tracer=tracer,
|
|
78
|
+
)
|
|
79
|
+
return client
|