metergraph 0.2.1__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {metergraph-0.2.1 → metergraph-0.3.0}/PKG-INFO +32 -12
- {metergraph-0.2.1 → metergraph-0.3.0}/README.md +31 -11
- {metergraph-0.2.1 → metergraph-0.3.0}/pyproject.toml +1 -1
- {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/__init__.py +14 -3
- {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/_capture.py +195 -14
- {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/_context.py +89 -0
- {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/_template.py +18 -2
- {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/_version.py +1 -1
- {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph.egg-info/PKG-INFO +32 -12
- {metergraph-0.2.1 → metergraph-0.3.0}/tests/test_sdk.py +210 -14
- {metergraph-0.2.1 → metergraph-0.3.0}/setup.cfg +0 -0
- {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/_config.py +0 -0
- {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/_failure_log.py +0 -0
- {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/_track.py +0 -0
- {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/_transport.py +0 -0
- {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph.egg-info/SOURCES.txt +0 -0
- {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph.egg-info/dependency_links.txt +0 -0
- {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph.egg-info/requires.txt +0 -0
- {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph.egg-info/top_level.txt +0 -0
- {metergraph-0.2.1 → metergraph-0.3.0}/tests/test_real_client_integration.py +0 -0
- {metergraph-0.2.1 → metergraph-0.3.0}/tests/test_seam_reality.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: metergraph
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Fire-and-forget LLM spend capture for Metergraph
|
|
5
5
|
Author: Pioneer Square Labs
|
|
6
6
|
License-Expression: Apache-2.0
|
|
@@ -34,9 +34,10 @@ from openai import OpenAI
|
|
|
34
34
|
client = metergraph.wrap(OpenAI())
|
|
35
35
|
metergraph.set_session("ticket-123")
|
|
36
36
|
|
|
37
|
-
with metergraph.
|
|
38
|
-
|
|
39
|
-
|
|
37
|
+
with metergraph.trace("ticket-workflow"):
|
|
38
|
+
with metergraph.route("ticket-classifier", unit="answer"):
|
|
39
|
+
model = metergraph.model_for("ticket-classifier", default="gpt-4.1-mini")
|
|
40
|
+
client.chat.completions.create(model=model, messages=[...])
|
|
40
41
|
|
|
41
42
|
# Emit this after the user-visible task resolves. It shares the bounded async
|
|
42
43
|
# transport and contains no prompt or output content.
|
|
@@ -54,15 +55,30 @@ Configuration:
|
|
|
54
55
|
|
|
55
56
|
- `METERGRAPH_APP_TOKEN` — required bearer token
|
|
56
57
|
- `METERGRAPH_INGEST_URL` — optional override; defaults to the hosted HTTPS endpoint
|
|
57
|
-
- `METERGRAPH_CAPTURE_TEXT=
|
|
58
|
+
- `METERGRAPH_CAPTURE_TEXT=0` — opt out of content capture globally
|
|
58
59
|
- `METERGRAPH_DISABLED=1` — process kill switch
|
|
59
60
|
- `METERGRAPH_QUEUE_SIZE`, `METERGRAPH_BATCH_SIZE`, `METERGRAPH_FLUSH_SECONDS`
|
|
60
61
|
|
|
61
62
|
Delivery is bounded and off the request path. Queue overflow or a collector
|
|
62
63
|
outage drops capture and increments internal counters; it never changes the
|
|
63
64
|
provider call. Each wire batch is bounded to 512 KiB after optional gzip.
|
|
64
|
-
|
|
65
|
-
|
|
65
|
+
SDK 0.3 captures the scrubbed provider request and a normalized response
|
|
66
|
+
envelope, including assistant content and tool calls, by default. Provider
|
|
67
|
+
credentials and transport headers are removed. Request and response are each
|
|
68
|
+
limited to 100 KiB of UTF-8 with an explicit truncation marker.
|
|
69
|
+
`capture_text=False` on `route()` or `trace()` overrides the global content
|
|
70
|
+
policy for a sensitive operation. The equivalent initialization option is
|
|
71
|
+
`metergraph.init(capture_text=False)`. The public open-source server continues
|
|
72
|
+
to discard content even when the SDK sends it; the hosted dashboard retains
|
|
73
|
+
content under the workspace retention period.
|
|
74
|
+
|
|
75
|
+
`metergraph.trace(name, trace_id=..., parent_span_id=...)` is a sync/async
|
|
76
|
+
context manager and decorator. Calls inside one trace share a trace ID and
|
|
77
|
+
receive distinct span IDs. Calls outside a trace become deterministic
|
|
78
|
+
single-span traces after ingestion. Manual IDs can join work across process
|
|
79
|
+
boundaries; automatic W3C HTTP propagation is not included.
|
|
80
|
+
|
|
81
|
+
Config reads are ETag-aware and fail open to the default model.
|
|
66
82
|
`record_outcome` requires a stable session ID and the model actually used so a
|
|
67
83
|
session-sticky canary can compare task completion and optional feedback,
|
|
68
84
|
turn-count, escalation, abandonment, edit-distance, and regeneration signals.
|
|
@@ -88,11 +104,15 @@ genai.Client() construction in place, e.g. client = metergraph.wrap(OpenAI()).
|
|
|
88
104
|
wrap() returns the same client and initializes itself from the environment:
|
|
89
105
|
METERGRAPH_APP_TOKEN is required (capture is silently off without it) and
|
|
90
106
|
METERGRAPH_INGEST_URL is only for self-hosted servers. Add both to
|
|
91
|
-
.env.example, and never commit a real token.
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
107
|
+
.env.example, and never commit a real token. SDK 0.3 captures scrubbed provider
|
|
108
|
+
requests and normalized responses by default for the hosted dashboard; use
|
|
109
|
+
METERGRAPH_CAPTURE_TEXT=0 or capture_text=False around sensitive operations.
|
|
110
|
+
Provider credentials and transport headers must never be captured. Capture is
|
|
111
|
+
fail-open, so do not change call sites, arguments, or error handling; sync,
|
|
112
|
+
async, and streaming work unchanged. Use metergraph.trace("stable-name") to
|
|
113
|
+
group multi-call workflows. Attribution to the calling function is automatic;
|
|
114
|
+
optionally pin stable names on key LLM-calling functions with
|
|
115
|
+
@metergraph.track. On
|
|
96
116
|
serverless, call metergraph.flush() before the handler returns. When done,
|
|
97
117
|
list every client you wrapped and flag LLM calls made outside the official
|
|
98
118
|
openai / anthropic / google-genai SDKs, since those are not captured.
|
|
@@ -13,9 +13,10 @@ from openai import OpenAI
|
|
|
13
13
|
client = metergraph.wrap(OpenAI())
|
|
14
14
|
metergraph.set_session("ticket-123")
|
|
15
15
|
|
|
16
|
-
with metergraph.
|
|
17
|
-
|
|
18
|
-
|
|
16
|
+
with metergraph.trace("ticket-workflow"):
|
|
17
|
+
with metergraph.route("ticket-classifier", unit="answer"):
|
|
18
|
+
model = metergraph.model_for("ticket-classifier", default="gpt-4.1-mini")
|
|
19
|
+
client.chat.completions.create(model=model, messages=[...])
|
|
19
20
|
|
|
20
21
|
# Emit this after the user-visible task resolves. It shares the bounded async
|
|
21
22
|
# transport and contains no prompt or output content.
|
|
@@ -33,15 +34,30 @@ Configuration:
|
|
|
33
34
|
|
|
34
35
|
- `METERGRAPH_APP_TOKEN` — required bearer token
|
|
35
36
|
- `METERGRAPH_INGEST_URL` — optional override; defaults to the hosted HTTPS endpoint
|
|
36
|
-
- `METERGRAPH_CAPTURE_TEXT=
|
|
37
|
+
- `METERGRAPH_CAPTURE_TEXT=0` — opt out of content capture globally
|
|
37
38
|
- `METERGRAPH_DISABLED=1` — process kill switch
|
|
38
39
|
- `METERGRAPH_QUEUE_SIZE`, `METERGRAPH_BATCH_SIZE`, `METERGRAPH_FLUSH_SECONDS`
|
|
39
40
|
|
|
40
41
|
Delivery is bounded and off the request path. Queue overflow or a collector
|
|
41
42
|
outage drops capture and increments internal counters; it never changes the
|
|
42
43
|
provider call. Each wire batch is bounded to 512 KiB after optional gzip.
|
|
43
|
-
|
|
44
|
-
|
|
44
|
+
SDK 0.3 captures the scrubbed provider request and a normalized response
|
|
45
|
+
envelope, including assistant content and tool calls, by default. Provider
|
|
46
|
+
credentials and transport headers are removed. Request and response are each
|
|
47
|
+
limited to 100 KiB of UTF-8 with an explicit truncation marker.
|
|
48
|
+
`capture_text=False` on `route()` or `trace()` overrides the global content
|
|
49
|
+
policy for a sensitive operation. The equivalent initialization option is
|
|
50
|
+
`metergraph.init(capture_text=False)`. The public open-source server continues
|
|
51
|
+
to discard content even when the SDK sends it; the hosted dashboard retains
|
|
52
|
+
content under the workspace retention period.
|
|
53
|
+
|
|
54
|
+
`metergraph.trace(name, trace_id=..., parent_span_id=...)` is a sync/async
|
|
55
|
+
context manager and decorator. Calls inside one trace share a trace ID and
|
|
56
|
+
receive distinct span IDs. Calls outside a trace become deterministic
|
|
57
|
+
single-span traces after ingestion. Manual IDs can join work across process
|
|
58
|
+
boundaries; automatic W3C HTTP propagation is not included.
|
|
59
|
+
|
|
60
|
+
Config reads are ETag-aware and fail open to the default model.
|
|
45
61
|
`record_outcome` requires a stable session ID and the model actually used so a
|
|
46
62
|
session-sticky canary can compare task completion and optional feedback,
|
|
47
63
|
turn-count, escalation, abandonment, edit-distance, and regeneration signals.
|
|
@@ -67,11 +83,15 @@ genai.Client() construction in place, e.g. client = metergraph.wrap(OpenAI()).
|
|
|
67
83
|
wrap() returns the same client and initializes itself from the environment:
|
|
68
84
|
METERGRAPH_APP_TOKEN is required (capture is silently off without it) and
|
|
69
85
|
METERGRAPH_INGEST_URL is only for self-hosted servers. Add both to
|
|
70
|
-
.env.example, and never commit a real token.
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
86
|
+
.env.example, and never commit a real token. SDK 0.3 captures scrubbed provider
|
|
87
|
+
requests and normalized responses by default for the hosted dashboard; use
|
|
88
|
+
METERGRAPH_CAPTURE_TEXT=0 or capture_text=False around sensitive operations.
|
|
89
|
+
Provider credentials and transport headers must never be captured. Capture is
|
|
90
|
+
fail-open, so do not change call sites, arguments, or error handling; sync,
|
|
91
|
+
async, and streaming work unchanged. Use metergraph.trace("stable-name") to
|
|
92
|
+
group multi-call workflows. Attribution to the calling function is automatic;
|
|
93
|
+
optionally pin stable names on key LLM-calling functions with
|
|
94
|
+
@metergraph.track. On
|
|
75
95
|
serverless, call metergraph.flush() before the handler returns. When done,
|
|
76
96
|
list every client you wrapped and flag LLM calls made outside the official
|
|
77
97
|
openai / anthropic / google-genai SDKs, since those are not captured.
|
|
@@ -12,7 +12,7 @@ from typing import Any, Callable
|
|
|
12
12
|
from ._capture import Options, Runtime, set_runtime
|
|
13
13
|
from ._capture import wrap as _wrap
|
|
14
14
|
from ._config import ConfigPoller
|
|
15
|
-
from ._context import route, set_session, set_tags, snapshot, wrap_executor
|
|
15
|
+
from ._context import route, set_session, set_tags, snapshot, trace, wrap_executor
|
|
16
16
|
from ._track import track
|
|
17
17
|
from ._transport import Writer
|
|
18
18
|
from ._version import SDK_VERSION
|
|
@@ -73,7 +73,7 @@ def init(
|
|
|
73
73
|
)
|
|
74
74
|
options = Options(
|
|
75
75
|
capture_text=(
|
|
76
|
-
_env_bool("METERGRAPH_CAPTURE_TEXT",
|
|
76
|
+
_env_bool("METERGRAPH_CAPTURE_TEXT", True)
|
|
77
77
|
if capture_text is None
|
|
78
78
|
else capture_text
|
|
79
79
|
),
|
|
@@ -81,7 +81,17 @@ def init(
|
|
|
81
81
|
app_root=os.path.realpath(app_root or os.getcwd()),
|
|
82
82
|
skip_frames=tuple(skip_frames or ()),
|
|
83
83
|
environment=environment or os.getenv("METERGRAPH_ENV"),
|
|
84
|
-
text_max_bytes=
|
|
84
|
+
text_max_bytes=min(
|
|
85
|
+
100 * 1024,
|
|
86
|
+
max(
|
|
87
|
+
1,
|
|
88
|
+
int(
|
|
89
|
+
os.getenv(
|
|
90
|
+
"METERGRAPH_TEXT_MAX_BYTES", str(100 * 1024)
|
|
91
|
+
)
|
|
92
|
+
),
|
|
93
|
+
),
|
|
94
|
+
),
|
|
85
95
|
)
|
|
86
96
|
set_runtime(Runtime(_writer, options))
|
|
87
97
|
_config = ConfigPoller(
|
|
@@ -214,6 +224,7 @@ __all__ = [
|
|
|
214
224
|
"set_tags",
|
|
215
225
|
"shutdown",
|
|
216
226
|
"track",
|
|
227
|
+
"trace",
|
|
217
228
|
"wrap",
|
|
218
229
|
"wrap_executor",
|
|
219
230
|
]
|
|
@@ -8,6 +8,7 @@ import json
|
|
|
8
8
|
import logging
|
|
9
9
|
import os
|
|
10
10
|
import platform
|
|
11
|
+
import secrets
|
|
11
12
|
import sys
|
|
12
13
|
import time
|
|
13
14
|
from dataclasses import dataclass
|
|
@@ -223,11 +224,68 @@ def _request_id(response: Any) -> str | None:
|
|
|
223
224
|
value = (
|
|
224
225
|
_get(response, "_request_id")
|
|
225
226
|
or _get(response, "response_id")
|
|
227
|
+
or _get(response, "responseId")
|
|
226
228
|
or _get(response, "id")
|
|
227
229
|
)
|
|
228
230
|
return str(value) if value is not None else None
|
|
229
231
|
|
|
230
232
|
|
|
233
|
+
def _response_content(response: Any, aggregate_text: str | None = None) -> Any:
|
|
234
|
+
if aggregate_text is not None:
|
|
235
|
+
return aggregate_text
|
|
236
|
+
direct = _get(response, "output_text") or _get(response, "text")
|
|
237
|
+
if direct is not None:
|
|
238
|
+
return scrub(direct)
|
|
239
|
+
choice = _first(_get(response, "choices"))
|
|
240
|
+
message = _get(choice, "message")
|
|
241
|
+
content = _get(message, "content")
|
|
242
|
+
if content is not None:
|
|
243
|
+
return scrub(content)
|
|
244
|
+
parsed = _get(message, "parsed")
|
|
245
|
+
if parsed is not None:
|
|
246
|
+
return scrub(parsed)
|
|
247
|
+
normalized_text = _response_text(response)
|
|
248
|
+
if normalized_text is not None:
|
|
249
|
+
return normalized_text
|
|
250
|
+
blocks = _get(response, "content")
|
|
251
|
+
if blocks is not None:
|
|
252
|
+
return scrub(blocks)
|
|
253
|
+
outputs = _get(response, "output")
|
|
254
|
+
if outputs is not None:
|
|
255
|
+
return scrub(outputs)
|
|
256
|
+
candidates = _get(response, "candidates")
|
|
257
|
+
if candidates is not None:
|
|
258
|
+
return scrub(candidates)
|
|
259
|
+
return None
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _response_envelope(
|
|
263
|
+
response: Any,
|
|
264
|
+
*,
|
|
265
|
+
aggregate_text: str | None,
|
|
266
|
+
tool_calls: list[dict] | None,
|
|
267
|
+
error: BaseException | None,
|
|
268
|
+
status: str,
|
|
269
|
+
) -> dict[str, Any]:
|
|
270
|
+
envelope: dict[str, Any] = {
|
|
271
|
+
"role": "assistant",
|
|
272
|
+
"content": _response_content(response, aggregate_text),
|
|
273
|
+
"tool_calls": tool_calls or [],
|
|
274
|
+
"finish_reason": _stop_reason(response),
|
|
275
|
+
"request_id": _request_id(response),
|
|
276
|
+
"model": _get(response, "model")
|
|
277
|
+
or _get(response, "model_version")
|
|
278
|
+
or _get(response, "modelVersion"),
|
|
279
|
+
"status": status,
|
|
280
|
+
}
|
|
281
|
+
if error is not None:
|
|
282
|
+
envelope["error"] = {
|
|
283
|
+
"type": type(error).__name__,
|
|
284
|
+
"message": str(error),
|
|
285
|
+
}
|
|
286
|
+
return {key: value for key, value in envelope.items() if value is not None}
|
|
287
|
+
|
|
288
|
+
|
|
231
289
|
def _tool_names(request: Mapping[str, Any]) -> list[dict[str, str]] | None:
|
|
232
290
|
tools = request.get("tools")
|
|
233
291
|
if not isinstance(tools, list):
|
|
@@ -273,7 +331,11 @@ def _tool_policies(request: Mapping[str, Any]) -> dict[str, str]:
|
|
|
273
331
|
return policies
|
|
274
332
|
|
|
275
333
|
|
|
276
|
-
def _tool_events(
|
|
334
|
+
def _tool_events(
|
|
335
|
+
request: Mapping[str, Any],
|
|
336
|
+
response: Any,
|
|
337
|
+
stream_chunks: list[Any] | None = None,
|
|
338
|
+
) -> list[dict] | None:
|
|
277
339
|
"""Normalize completed history and newly requested provider tool calls."""
|
|
278
340
|
policies = _tool_policies(request)
|
|
279
341
|
calls: dict[str, dict] = {}
|
|
@@ -320,9 +382,37 @@ def _tool_events(request: Mapping[str, Any], response: Any) -> list[dict] | None
|
|
|
320
382
|
bool(_get(block, "is_error", False)),
|
|
321
383
|
)
|
|
322
384
|
|
|
385
|
+
def gemini_parts(value: Any) -> None:
|
|
386
|
+
parts = _get(value, "parts")
|
|
387
|
+
if not isinstance(parts, list):
|
|
388
|
+
return
|
|
389
|
+
for part in parts:
|
|
390
|
+
function_call = _get(part, "function_call") or _get(
|
|
391
|
+
part, "functionCall"
|
|
392
|
+
)
|
|
393
|
+
if function_call is not None:
|
|
394
|
+
call(
|
|
395
|
+
_get(function_call, "id"),
|
|
396
|
+
_get(function_call, "name"),
|
|
397
|
+
_get(function_call, "args")
|
|
398
|
+
or _get(function_call, "arguments"),
|
|
399
|
+
)
|
|
400
|
+
function_response = _get(part, "function_response") or _get(
|
|
401
|
+
part, "functionResponse"
|
|
402
|
+
)
|
|
403
|
+
if function_response is not None:
|
|
404
|
+
complete(
|
|
405
|
+
_get(function_response, "id")
|
|
406
|
+
or _get(function_response, "name"),
|
|
407
|
+
_get(function_response, "response"),
|
|
408
|
+
False,
|
|
409
|
+
)
|
|
410
|
+
|
|
323
411
|
history = request.get("messages")
|
|
324
412
|
if not isinstance(history, list):
|
|
325
413
|
history = request.get("input")
|
|
414
|
+
if not isinstance(history, list):
|
|
415
|
+
history = request.get("contents")
|
|
326
416
|
if isinstance(history, list):
|
|
327
417
|
for message in history:
|
|
328
418
|
for tool_call in _get(message, "tool_calls", []) or []:
|
|
@@ -353,6 +443,7 @@ def _tool_events(request: Mapping[str, Any], response: Any) -> list[dict] | None
|
|
|
353
443
|
bool(_get(message, "is_error", False)),
|
|
354
444
|
)
|
|
355
445
|
content_blocks(_get(message, "content"))
|
|
446
|
+
gemini_parts(message)
|
|
356
447
|
|
|
357
448
|
choice = _first(_get(response, "choices"))
|
|
358
449
|
response_message = _get(choice, "message")
|
|
@@ -372,6 +463,60 @@ def _tool_events(request: Mapping[str, Any], response: Any) -> list[dict] | None
|
|
|
372
463
|
_get(output, "name"),
|
|
373
464
|
_get(output, "arguments"),
|
|
374
465
|
)
|
|
466
|
+
for candidate in _get(response, "candidates", []) or []:
|
|
467
|
+
gemini_parts(_get(candidate, "content"))
|
|
468
|
+
|
|
469
|
+
openai_deltas: dict[str, dict[str, str]] = {}
|
|
470
|
+
anthropic_deltas: dict[str, dict[str, str]] = {}
|
|
471
|
+
for chunk in stream_chunks or []:
|
|
472
|
+
for choice in _get(chunk, "choices", []) or []:
|
|
473
|
+
delta = _get(choice, "delta")
|
|
474
|
+
for position, tool_call in enumerate(
|
|
475
|
+
_get(delta, "tool_calls", []) or []
|
|
476
|
+
):
|
|
477
|
+
key = str(
|
|
478
|
+
_get(tool_call, "index")
|
|
479
|
+
if _get(tool_call, "index") is not None
|
|
480
|
+
else position
|
|
481
|
+
)
|
|
482
|
+
item = openai_deltas.setdefault(
|
|
483
|
+
key, {"id": key, "name": "", "arguments": ""}
|
|
484
|
+
)
|
|
485
|
+
if _get(tool_call, "id"):
|
|
486
|
+
item["id"] = str(_get(tool_call, "id"))
|
|
487
|
+
fn = _get(tool_call, "function")
|
|
488
|
+
if _get(fn, "name"):
|
|
489
|
+
item["name"] += str(_get(fn, "name"))
|
|
490
|
+
if _get(fn, "arguments"):
|
|
491
|
+
item["arguments"] += str(_get(fn, "arguments"))
|
|
492
|
+
kind = _get(chunk, "type")
|
|
493
|
+
if kind == "content_block_start":
|
|
494
|
+
block = _get(chunk, "content_block")
|
|
495
|
+
if _get(block, "type") == "tool_use":
|
|
496
|
+
key = str(_get(chunk, "index", len(anthropic_deltas)))
|
|
497
|
+
initial_input = scrub(_get(block, "input", {}))
|
|
498
|
+
anthropic_deltas[key] = {
|
|
499
|
+
"id": str(_get(block, "id") or key),
|
|
500
|
+
"name": str(_get(block, "name") or ""),
|
|
501
|
+
"arguments": (
|
|
502
|
+
""
|
|
503
|
+
if initial_input == {}
|
|
504
|
+
else json.dumps(initial_input, separators=(",", ":"))
|
|
505
|
+
),
|
|
506
|
+
}
|
|
507
|
+
elif kind == "content_block_delta":
|
|
508
|
+
delta = _get(chunk, "delta")
|
|
509
|
+
if _get(delta, "type") == "input_json_delta":
|
|
510
|
+
key = str(_get(chunk, "index", "0"))
|
|
511
|
+
item = anthropic_deltas.setdefault(
|
|
512
|
+
key, {"id": key, "name": "", "arguments": ""}
|
|
513
|
+
)
|
|
514
|
+
item["arguments"] += str(_get(delta, "partial_json") or "")
|
|
515
|
+
for candidate in _get(chunk, "candidates", []) or []:
|
|
516
|
+
gemini_parts(_get(candidate, "content"))
|
|
517
|
+
for item in [*openai_deltas.values(), *anthropic_deltas.values()]:
|
|
518
|
+
if item["name"]:
|
|
519
|
+
call(item["id"], item["name"], item["arguments"])
|
|
375
520
|
|
|
376
521
|
return [calls[key] for key in order] or None
|
|
377
522
|
|
|
@@ -399,12 +544,12 @@ def _capture_frames(
|
|
|
399
544
|
|
|
400
545
|
@dataclass
|
|
401
546
|
class Options:
|
|
402
|
-
capture_text: bool =
|
|
547
|
+
capture_text: bool = True
|
|
403
548
|
redact: Callable[[str, str], str] | None = None
|
|
404
549
|
app_root: str = os.getcwd()
|
|
405
550
|
skip_frames: tuple[str, ...] = ()
|
|
406
551
|
environment: str | None = None
|
|
407
|
-
text_max_bytes: int =
|
|
552
|
+
text_max_bytes: int = 100 * 1024
|
|
408
553
|
|
|
409
554
|
|
|
410
555
|
class Runtime:
|
|
@@ -442,6 +587,14 @@ class Runtime:
|
|
|
442
587
|
func=context.func_name or func,
|
|
443
588
|
module=context.func_module or module,
|
|
444
589
|
frames=frames,
|
|
590
|
+
trace_id=context.trace_id or secrets.token_hex(16),
|
|
591
|
+
span_id=secrets.token_hex(8),
|
|
592
|
+
parent_span_id=context.parent_span_id,
|
|
593
|
+
trace_name=context.trace_name
|
|
594
|
+
or context.route
|
|
595
|
+
or context.func_name
|
|
596
|
+
or func
|
|
597
|
+
or endpoint,
|
|
445
598
|
)
|
|
446
599
|
|
|
447
600
|
def _text(
|
|
@@ -476,6 +629,10 @@ class CallState:
|
|
|
476
629
|
func: str | None
|
|
477
630
|
module: str | None
|
|
478
631
|
frames: list[dict]
|
|
632
|
+
trace_id: str
|
|
633
|
+
span_id: str
|
|
634
|
+
parent_span_id: str | None
|
|
635
|
+
trace_name: str
|
|
479
636
|
done: bool = False
|
|
480
637
|
|
|
481
638
|
def finish(
|
|
@@ -487,6 +644,7 @@ class CallState:
|
|
|
487
644
|
stream: bool = False,
|
|
488
645
|
ttft_ms: int | None = None,
|
|
489
646
|
response_text: str | None = None,
|
|
647
|
+
stream_chunks: list[Any] | None = None,
|
|
490
648
|
) -> None:
|
|
491
649
|
if self.done:
|
|
492
650
|
return
|
|
@@ -502,17 +660,13 @@ class CallState:
|
|
|
502
660
|
"request",
|
|
503
661
|
enabled=capture_text,
|
|
504
662
|
)
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
"response",
|
|
508
|
-
enabled=capture_text,
|
|
509
|
-
)
|
|
510
|
-
tool_calls = _tool_events(request_clean, response)
|
|
663
|
+
full_tool_calls = _tool_events(request_clean, response, stream_chunks)
|
|
664
|
+
tool_calls = full_tool_calls
|
|
511
665
|
tool_truncated = False
|
|
512
666
|
if tool_calls and capture_text:
|
|
513
667
|
encoded_tools, tool_truncated = self.runtime._text(
|
|
514
668
|
json.dumps(tool_calls, separators=(",", ":"), default=repr),
|
|
515
|
-
"
|
|
669
|
+
"response",
|
|
516
670
|
enabled=True,
|
|
517
671
|
)
|
|
518
672
|
try:
|
|
@@ -529,6 +683,25 @@ class CallState:
|
|
|
529
683
|
}
|
|
530
684
|
for item in tool_calls
|
|
531
685
|
]
|
|
686
|
+
effective_status = status or (
|
|
687
|
+
"error" if error else _stop_reason(response) or "success"
|
|
688
|
+
)
|
|
689
|
+
response_json, response_truncated = self.runtime._text(
|
|
690
|
+
json.dumps(
|
|
691
|
+
_response_envelope(
|
|
692
|
+
response,
|
|
693
|
+
aggregate_text=response_text,
|
|
694
|
+
tool_calls=full_tool_calls if capture_text else None,
|
|
695
|
+
error=error,
|
|
696
|
+
status=effective_status,
|
|
697
|
+
),
|
|
698
|
+
ensure_ascii=False,
|
|
699
|
+
separators=(",", ":"),
|
|
700
|
+
default=repr,
|
|
701
|
+
),
|
|
702
|
+
"response",
|
|
703
|
+
enabled=capture_text,
|
|
704
|
+
)
|
|
532
705
|
row: dict[str, Any] = {
|
|
533
706
|
"ts": self.ts,
|
|
534
707
|
"route": self.context.route,
|
|
@@ -536,10 +709,13 @@ class CallState:
|
|
|
536
709
|
"model": self.request.get("model"),
|
|
537
710
|
**_usage(response),
|
|
538
711
|
"latency_ms": round((time.perf_counter() - self.started) * 1000),
|
|
539
|
-
"status":
|
|
540
|
-
or ("error" if error else _stop_reason(response) or "success"),
|
|
712
|
+
"status": effective_status,
|
|
541
713
|
"session_id": self.context.session_id,
|
|
542
714
|
"conversation_id": self.context.session_id,
|
|
715
|
+
"trace_id": self.trace_id,
|
|
716
|
+
"span_id": self.span_id,
|
|
717
|
+
"parent_span_id": self.parent_span_id,
|
|
718
|
+
"trace_name": self.trace_name,
|
|
543
719
|
"template_hash": template_hash(self.request),
|
|
544
720
|
"unit_name": self.context.unit_name,
|
|
545
721
|
"unit_count": self.context.unit_count,
|
|
@@ -548,8 +724,8 @@ class CallState:
|
|
|
548
724
|
"request_id": _request_id(response),
|
|
549
725
|
"batch": self.request.get("batch") is True,
|
|
550
726
|
"batch_custom_id": self.request.get("batch_custom_id"),
|
|
551
|
-
#
|
|
552
|
-
#
|
|
727
|
+
# An explicit false is the application-level sensitive-operation
|
|
728
|
+
# opt-out. Hosted ingestion otherwise preserves available content.
|
|
553
729
|
"content_opted_in": capture_text,
|
|
554
730
|
"request_json": request_json,
|
|
555
731
|
"response_text": response_json,
|
|
@@ -580,10 +756,12 @@ class _StreamState:
|
|
|
580
756
|
self.iterator = None
|
|
581
757
|
self.last = None
|
|
582
758
|
self.parts: list[str] = []
|
|
759
|
+
self.chunks: list[Any] = []
|
|
583
760
|
self.ttft_ms: int | None = None
|
|
584
761
|
|
|
585
762
|
def chunk(self, value: Any) -> Any:
|
|
586
763
|
self.last = value
|
|
764
|
+
self.chunks.append(value)
|
|
587
765
|
text = _chunk_text(value)
|
|
588
766
|
if text:
|
|
589
767
|
if self.ttft_ms is None:
|
|
@@ -614,6 +792,7 @@ class _StreamState:
|
|
|
614
792
|
stream=True,
|
|
615
793
|
ttft_ms=self.ttft_ms,
|
|
616
794
|
response_text="".join(self.parts) or None,
|
|
795
|
+
stream_chunks=self.chunks,
|
|
617
796
|
)
|
|
618
797
|
|
|
619
798
|
async def finish_async(
|
|
@@ -636,6 +815,7 @@ class _StreamState:
|
|
|
636
815
|
stream=True,
|
|
637
816
|
ttft_ms=self.ttft_ms,
|
|
638
817
|
response_text="".join(self.parts) or None,
|
|
818
|
+
stream_chunks=self.chunks,
|
|
639
819
|
)
|
|
640
820
|
|
|
641
821
|
|
|
@@ -754,6 +934,7 @@ class AsyncStream:
|
|
|
754
934
|
stream=True,
|
|
755
935
|
ttft_ms=self._state.ttft_ms,
|
|
756
936
|
response_text="".join(self._state.parts) or None,
|
|
937
|
+
stream_chunks=self._state.chunks,
|
|
757
938
|
)
|
|
758
939
|
except Exception:
|
|
759
940
|
pass
|
|
@@ -5,6 +5,7 @@ from __future__ import annotations
|
|
|
5
5
|
import contextvars
|
|
6
6
|
import functools
|
|
7
7
|
import inspect
|
|
8
|
+
import secrets
|
|
8
9
|
from concurrent.futures import Executor
|
|
9
10
|
from dataclasses import dataclass, field, replace
|
|
10
11
|
from typing import Any, Callable, Mapping
|
|
@@ -20,6 +21,9 @@ class CaptureContext:
|
|
|
20
21
|
capture_text: bool | None = None
|
|
21
22
|
func_name: str | None = None
|
|
22
23
|
func_module: str | None = None
|
|
24
|
+
trace_id: str | None = None
|
|
25
|
+
trace_name: str | None = None
|
|
26
|
+
parent_span_id: str | None = None
|
|
23
27
|
|
|
24
28
|
|
|
25
29
|
_current: contextvars.ContextVar[CaptureContext] = contextvars.ContextVar(
|
|
@@ -108,6 +112,91 @@ class route:
|
|
|
108
112
|
return wrapped
|
|
109
113
|
|
|
110
114
|
|
|
115
|
+
class trace:
|
|
116
|
+
"""Logical trace context manager and sync/async decorator."""
|
|
117
|
+
|
|
118
|
+
def __init__(
|
|
119
|
+
self,
|
|
120
|
+
name: str,
|
|
121
|
+
*,
|
|
122
|
+
trace_id: str | None = None,
|
|
123
|
+
parent_span_id: str | None = None,
|
|
124
|
+
capture_text: bool | None = None,
|
|
125
|
+
) -> None:
|
|
126
|
+
self.name = str(name)
|
|
127
|
+
self.trace_id = str(trace_id).strip() if trace_id is not None else None
|
|
128
|
+
self.parent_span_id = (
|
|
129
|
+
str(parent_span_id).strip() if parent_span_id is not None else None
|
|
130
|
+
)
|
|
131
|
+
self.capture_text = (
|
|
132
|
+
bool(capture_text) if capture_text is not None else None
|
|
133
|
+
)
|
|
134
|
+
self._token: contextvars.Token[CaptureContext] | None = None
|
|
135
|
+
|
|
136
|
+
def __enter__(self) -> "trace":
|
|
137
|
+
current = snapshot()
|
|
138
|
+
requested = self.trace_id
|
|
139
|
+
reuse = current.trace_id is not None and (
|
|
140
|
+
requested is None or requested == current.trace_id
|
|
141
|
+
)
|
|
142
|
+
self._token = _current.set(
|
|
143
|
+
replace(
|
|
144
|
+
current,
|
|
145
|
+
trace_id=(
|
|
146
|
+
current.trace_id
|
|
147
|
+
if reuse
|
|
148
|
+
else requested or secrets.token_hex(16)
|
|
149
|
+
),
|
|
150
|
+
trace_name=current.trace_name if reuse else self.name,
|
|
151
|
+
parent_span_id=(
|
|
152
|
+
self.parent_span_id
|
|
153
|
+
if self.parent_span_id is not None
|
|
154
|
+
else current.parent_span_id
|
|
155
|
+
if reuse
|
|
156
|
+
else None
|
|
157
|
+
),
|
|
158
|
+
capture_text=(
|
|
159
|
+
self.capture_text
|
|
160
|
+
if self.capture_text is not None
|
|
161
|
+
else current.capture_text
|
|
162
|
+
),
|
|
163
|
+
)
|
|
164
|
+
)
|
|
165
|
+
return self
|
|
166
|
+
|
|
167
|
+
def __exit__(self, exc_type, exc, tb) -> None:
|
|
168
|
+
if self._token is not None:
|
|
169
|
+
_current.reset(self._token)
|
|
170
|
+
self._token = None
|
|
171
|
+
|
|
172
|
+
def __call__(self, fn: Callable):
|
|
173
|
+
if inspect.iscoroutinefunction(fn):
|
|
174
|
+
|
|
175
|
+
@functools.wraps(fn)
|
|
176
|
+
async def async_wrapped(*args, **kwargs):
|
|
177
|
+
with type(self)(
|
|
178
|
+
self.name,
|
|
179
|
+
trace_id=self.trace_id,
|
|
180
|
+
parent_span_id=self.parent_span_id,
|
|
181
|
+
capture_text=self.capture_text,
|
|
182
|
+
):
|
|
183
|
+
return await fn(*args, **kwargs)
|
|
184
|
+
|
|
185
|
+
return async_wrapped
|
|
186
|
+
|
|
187
|
+
@functools.wraps(fn)
|
|
188
|
+
def wrapped(*args, **kwargs):
|
|
189
|
+
with type(self)(
|
|
190
|
+
self.name,
|
|
191
|
+
trace_id=self.trace_id,
|
|
192
|
+
parent_span_id=self.parent_span_id,
|
|
193
|
+
capture_text=self.capture_text,
|
|
194
|
+
):
|
|
195
|
+
return fn(*args, **kwargs)
|
|
196
|
+
|
|
197
|
+
return wrapped
|
|
198
|
+
|
|
199
|
+
|
|
111
200
|
def set_session(session_id: str | None) -> None:
|
|
112
201
|
_current.set(
|
|
113
202
|
replace(snapshot(), session_id=str(session_id) if session_id else None)
|
|
@@ -14,7 +14,23 @@ _EMAIL = re.compile(r"\b[^\s@]+@[^\s@]+\.[^\s@]+\b")
|
|
|
14
14
|
_URL = re.compile(r"\bhttps?://\S+")
|
|
15
15
|
_NUMBER = re.compile(r"(?<![A-Za-z])[-+]?\d+(?:\.\d+)?(?![A-Za-z])")
|
|
16
16
|
_LONG_TOKEN = re.compile(r"\b[A-Za-z0-9_-]{24,}\b")
|
|
17
|
-
_SENSITIVE_KEYS = {
|
|
17
|
+
_SENSITIVE_KEYS = {
|
|
18
|
+
"api-key",
|
|
19
|
+
"api_key",
|
|
20
|
+
"apikey",
|
|
21
|
+
"authorization",
|
|
22
|
+
"client_secret",
|
|
23
|
+
"cookie",
|
|
24
|
+
"headers",
|
|
25
|
+
"id_token",
|
|
26
|
+
"password",
|
|
27
|
+
"proxy-authorization",
|
|
28
|
+
"refresh_token",
|
|
29
|
+
"secret",
|
|
30
|
+
"set-cookie",
|
|
31
|
+
"token",
|
|
32
|
+
"x-api-key",
|
|
33
|
+
}
|
|
18
34
|
|
|
19
35
|
|
|
20
36
|
def _normalize_text(value: str) -> str:
|
|
@@ -31,7 +47,7 @@ def scrub(value: Any) -> Any:
|
|
|
31
47
|
return {
|
|
32
48
|
str(k): scrub(v)
|
|
33
49
|
for k, v in value.items()
|
|
34
|
-
if str(k).lower() not in _SENSITIVE_KEYS
|
|
50
|
+
if str(k).strip().lower() not in _SENSITIVE_KEYS
|
|
35
51
|
}
|
|
36
52
|
if isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)):
|
|
37
53
|
return [scrub(item) for item in value]
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: metergraph
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Fire-and-forget LLM spend capture for Metergraph
|
|
5
5
|
Author: Pioneer Square Labs
|
|
6
6
|
License-Expression: Apache-2.0
|
|
@@ -34,9 +34,10 @@ from openai import OpenAI
|
|
|
34
34
|
client = metergraph.wrap(OpenAI())
|
|
35
35
|
metergraph.set_session("ticket-123")
|
|
36
36
|
|
|
37
|
-
with metergraph.
|
|
38
|
-
|
|
39
|
-
|
|
37
|
+
with metergraph.trace("ticket-workflow"):
|
|
38
|
+
with metergraph.route("ticket-classifier", unit="answer"):
|
|
39
|
+
model = metergraph.model_for("ticket-classifier", default="gpt-4.1-mini")
|
|
40
|
+
client.chat.completions.create(model=model, messages=[...])
|
|
40
41
|
|
|
41
42
|
# Emit this after the user-visible task resolves. It shares the bounded async
|
|
42
43
|
# transport and contains no prompt or output content.
|
|
@@ -54,15 +55,30 @@ Configuration:
|
|
|
54
55
|
|
|
55
56
|
- `METERGRAPH_APP_TOKEN` — required bearer token
|
|
56
57
|
- `METERGRAPH_INGEST_URL` — optional override; defaults to the hosted HTTPS endpoint
|
|
57
|
-
- `METERGRAPH_CAPTURE_TEXT=
|
|
58
|
+
- `METERGRAPH_CAPTURE_TEXT=0` — opt out of content capture globally
|
|
58
59
|
- `METERGRAPH_DISABLED=1` — process kill switch
|
|
59
60
|
- `METERGRAPH_QUEUE_SIZE`, `METERGRAPH_BATCH_SIZE`, `METERGRAPH_FLUSH_SECONDS`
|
|
60
61
|
|
|
61
62
|
Delivery is bounded and off the request path. Queue overflow or a collector
|
|
62
63
|
outage drops capture and increments internal counters; it never changes the
|
|
63
64
|
provider call. Each wire batch is bounded to 512 KiB after optional gzip.
|
|
64
|
-
|
|
65
|
-
|
|
65
|
+
SDK 0.3 captures the scrubbed provider request and a normalized response
|
|
66
|
+
envelope, including assistant content and tool calls, by default. Provider
|
|
67
|
+
credentials and transport headers are removed. Request and response are each
|
|
68
|
+
limited to 100 KiB of UTF-8 with an explicit truncation marker.
|
|
69
|
+
`capture_text=False` on `route()` or `trace()` overrides the global content
|
|
70
|
+
policy for a sensitive operation. The equivalent initialization option is
|
|
71
|
+
`metergraph.init(capture_text=False)`. The public open-source server continues
|
|
72
|
+
to discard content even when the SDK sends it; the hosted dashboard retains
|
|
73
|
+
content under the workspace retention period.
|
|
74
|
+
|
|
75
|
+
`metergraph.trace(name, trace_id=..., parent_span_id=...)` is a sync/async
|
|
76
|
+
context manager and decorator. Calls inside one trace share a trace ID and
|
|
77
|
+
receive distinct span IDs. Calls outside a trace become deterministic
|
|
78
|
+
single-span traces after ingestion. Manual IDs can join work across process
|
|
79
|
+
boundaries; automatic W3C HTTP propagation is not included.
|
|
80
|
+
|
|
81
|
+
Config reads are ETag-aware and fail open to the default model.
|
|
66
82
|
`record_outcome` requires a stable session ID and the model actually used so a
|
|
67
83
|
session-sticky canary can compare task completion and optional feedback,
|
|
68
84
|
turn-count, escalation, abandonment, edit-distance, and regeneration signals.
|
|
@@ -88,11 +104,15 @@ genai.Client() construction in place, e.g. client = metergraph.wrap(OpenAI()).
|
|
|
88
104
|
wrap() returns the same client and initializes itself from the environment:
|
|
89
105
|
METERGRAPH_APP_TOKEN is required (capture is silently off without it) and
|
|
90
106
|
METERGRAPH_INGEST_URL is only for self-hosted servers. Add both to
|
|
91
|
-
.env.example, and never commit a real token.
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
107
|
+
.env.example, and never commit a real token. SDK 0.3 captures scrubbed provider
|
|
108
|
+
requests and normalized responses by default for the hosted dashboard; use
|
|
109
|
+
METERGRAPH_CAPTURE_TEXT=0 or capture_text=False around sensitive operations.
|
|
110
|
+
Provider credentials and transport headers must never be captured. Capture is
|
|
111
|
+
fail-open, so do not change call sites, arguments, or error handling; sync,
|
|
112
|
+
async, and streaming work unchanged. Use metergraph.trace("stable-name") to
|
|
113
|
+
group multi-call workflows. Attribution to the calling function is automatic;
|
|
114
|
+
optionally pin stable names on key LLM-calling functions with
|
|
115
|
+
@metergraph.track. On
|
|
96
116
|
serverless, call metergraph.flush() before the handler returns. When done,
|
|
97
117
|
list every client you wrapped and flag LLM calls made outside the official
|
|
98
118
|
openai / anthropic / google-genai SDKs, since those are not captured.
|
|
@@ -63,6 +63,10 @@ def response(text="done"):
|
|
|
63
63
|
)
|
|
64
64
|
|
|
65
65
|
|
|
66
|
+
def captured_response(row):
|
|
67
|
+
return json.loads(row["response_text"])
|
|
68
|
+
|
|
69
|
+
|
|
66
70
|
def test_wrap_auto_init_does_not_latch_before_a_token_is_available():
|
|
67
71
|
class Completions:
|
|
68
72
|
def create(self, **kwargs):
|
|
@@ -283,7 +287,7 @@ def test_wrap_google_records_usage_and_endpoint(tmp_path):
|
|
|
283
287
|
assert row["output_tokens"] == 20
|
|
284
288
|
assert row["cache_read_tokens"] == 10
|
|
285
289
|
assert row["reasoning_tokens"] == 5
|
|
286
|
-
assert row["
|
|
290
|
+
assert captured_response(row)["content"] == "gemini done"
|
|
287
291
|
assert row["request_id"] == "resp_g_1"
|
|
288
292
|
assert row["sdk_version"] == metergraph.__version__
|
|
289
293
|
_capture.set_runtime(None)
|
|
@@ -331,7 +335,7 @@ def test_wrap_google_stream_takes_usage_from_cumulative_last_chunk(tmp_path):
|
|
|
331
335
|
assert row["input_tokens"] == 100
|
|
332
336
|
assert row["output_tokens"] == 20
|
|
333
337
|
assert row["cache_read_tokens"] == 10
|
|
334
|
-
assert row["
|
|
338
|
+
assert captured_response(row)["content"] == "partial"
|
|
335
339
|
_capture.set_runtime(None)
|
|
336
340
|
|
|
337
341
|
|
|
@@ -376,11 +380,11 @@ def test_wrap_google_patches_async_models(tmp_path):
|
|
|
376
380
|
assert len(asyncio.run(run())) == 1
|
|
377
381
|
assert rows.rows[0]["provider"] == "google"
|
|
378
382
|
assert rows.rows[0]["endpoint"] == "models.generate_content"
|
|
379
|
-
assert rows.rows[0]["
|
|
383
|
+
assert captured_response(rows.rows[0])["content"] == "gemini async done"
|
|
380
384
|
assert rows.rows[1]["endpoint"] == "models.generate_content.stream"
|
|
381
385
|
assert rows.rows[1]["stream"] is True
|
|
382
386
|
assert rows.rows[1]["input_tokens"] == 100
|
|
383
|
-
assert rows.rows[1]["
|
|
387
|
+
assert captured_response(rows.rows[1])["content"] == "gemini async stream"
|
|
384
388
|
_capture.set_runtime(None)
|
|
385
389
|
|
|
386
390
|
|
|
@@ -550,6 +554,55 @@ def test_anthropic_response_tool_use_is_requested_not_replayable(tmp_path):
|
|
|
550
554
|
"idempotency": "non_idempotent",
|
|
551
555
|
}
|
|
552
556
|
]
|
|
557
|
+
assert captured_response(rows.rows[0])["tool_calls"] == rows.rows[0]["tool_calls"]
|
|
558
|
+
|
|
559
|
+
|
|
560
|
+
def test_gemini_response_normalizes_function_calls(tmp_path):
|
|
561
|
+
rows = Rows()
|
|
562
|
+
runtime = Runtime(rows, Options(app_root=str(tmp_path)))
|
|
563
|
+
call = runtime.call_state(
|
|
564
|
+
"google",
|
|
565
|
+
"models.generate_content",
|
|
566
|
+
{
|
|
567
|
+
"model": "gemini-test",
|
|
568
|
+
"contents": [{"role": "user", "parts": [{"text": "Find order"}]}],
|
|
569
|
+
"tools": [{"name": "lookup_order"}],
|
|
570
|
+
},
|
|
571
|
+
)
|
|
572
|
+
result = SimpleNamespace(
|
|
573
|
+
candidates=[
|
|
574
|
+
SimpleNamespace(
|
|
575
|
+
content=SimpleNamespace(
|
|
576
|
+
parts=[
|
|
577
|
+
SimpleNamespace(
|
|
578
|
+
functionCall=SimpleNamespace(
|
|
579
|
+
id="call_g_1",
|
|
580
|
+
name="lookup_order",
|
|
581
|
+
args={"order_id": "ord_g_1"},
|
|
582
|
+
)
|
|
583
|
+
)
|
|
584
|
+
]
|
|
585
|
+
)
|
|
586
|
+
)
|
|
587
|
+
],
|
|
588
|
+
usage_metadata=SimpleNamespace(
|
|
589
|
+
prompt_token_count=10, candidates_token_count=2
|
|
590
|
+
),
|
|
591
|
+
)
|
|
592
|
+
|
|
593
|
+
call.finish(result)
|
|
594
|
+
|
|
595
|
+
assert rows.rows[0]["tool_calls"] == [
|
|
596
|
+
{
|
|
597
|
+
"call_id": "call_g_1",
|
|
598
|
+
"name": "lookup_order",
|
|
599
|
+
"arguments": {"order_id": "ord_g_1"},
|
|
600
|
+
"result": None,
|
|
601
|
+
"status": "requested",
|
|
602
|
+
"idempotency": "non_idempotent",
|
|
603
|
+
}
|
|
604
|
+
]
|
|
605
|
+
assert captured_response(rows.rows[0])["tool_calls"] == rows.rows[0]["tool_calls"]
|
|
553
606
|
|
|
554
607
|
|
|
555
608
|
def test_openai_batch_output_file_captures_each_inference(tmp_path):
|
|
@@ -595,7 +648,7 @@ def test_openai_batch_output_file_captures_each_inference(tmp_path):
|
|
|
595
648
|
assert row["batch_custom_id"] == "ticket-1"
|
|
596
649
|
assert row["input_tokens"] == 11
|
|
597
650
|
assert row["output_tokens"] == 3
|
|
598
|
-
assert row["
|
|
651
|
+
assert captured_response(row)["content"] == "batch answer"
|
|
599
652
|
assert row["request_id"] == "req_batch_1"
|
|
600
653
|
|
|
601
654
|
# Re-reading the same output file in one process cannot double count it.
|
|
@@ -653,11 +706,11 @@ def test_anthropic_batch_results_capture_usage_without_changing_iteration(tmp_pa
|
|
|
653
706
|
assert row["cache_write_5m_tokens"] == 2
|
|
654
707
|
assert row["cache_write_1h_tokens"] == 5
|
|
655
708
|
assert "cost_usd" not in row
|
|
656
|
-
assert row["
|
|
709
|
+
assert captured_response(row)["content"] == "anthropic batch answer"
|
|
657
710
|
_capture.set_runtime(None)
|
|
658
711
|
|
|
659
712
|
|
|
660
|
-
def
|
|
713
|
+
def test_content_defaults_on_and_route_can_override_capture(tmp_path):
|
|
661
714
|
rows = Rows()
|
|
662
715
|
runtime = Runtime(rows, Options(app_root=str(tmp_path)))
|
|
663
716
|
call = runtime.call_state(
|
|
@@ -665,9 +718,9 @@ def test_content_defaults_off_and_route_can_override_global_consent(tmp_path):
|
|
|
665
718
|
)
|
|
666
719
|
call.finish(response("private output"))
|
|
667
720
|
|
|
668
|
-
assert rows.rows[0]["content_opted_in"] is
|
|
669
|
-
assert rows.rows[0]["request_json"]
|
|
670
|
-
assert rows.rows[0]["
|
|
721
|
+
assert rows.rows[0]["content_opted_in"] is True
|
|
722
|
+
assert "private" in rows.rows[0]["request_json"]
|
|
723
|
+
assert captured_response(rows.rows[0])["content"] == "private output"
|
|
671
724
|
|
|
672
725
|
_capture.set_runtime(runtime)
|
|
673
726
|
|
|
@@ -682,7 +735,7 @@ def test_content_defaults_off_and_route_can_override_global_consent(tmp_path):
|
|
|
682
735
|
|
|
683
736
|
assert rows.rows[1]["content_opted_in"] is True
|
|
684
737
|
assert "consented input" in rows.rows[1]["request_json"]
|
|
685
|
-
assert rows.rows[1]["
|
|
738
|
+
assert captured_response(rows.rows[1])["content"] == "consented output"
|
|
686
739
|
_capture.set_runtime(None)
|
|
687
740
|
|
|
688
741
|
|
|
@@ -706,6 +759,99 @@ def test_route_opt_out_overrides_global_content_capture(tmp_path):
|
|
|
706
759
|
_capture.set_runtime(None)
|
|
707
760
|
|
|
708
761
|
|
|
762
|
+
def test_trace_groups_spans_propagates_ids_and_supports_decorators(tmp_path):
|
|
763
|
+
rows = Rows()
|
|
764
|
+
runtime = Runtime(rows, Options(app_root=str(tmp_path)))
|
|
765
|
+
_capture.set_runtime(runtime)
|
|
766
|
+
|
|
767
|
+
class Responses:
|
|
768
|
+
def create(self, **kwargs):
|
|
769
|
+
return response(kwargs["input"])
|
|
770
|
+
|
|
771
|
+
client = SimpleNamespace(responses=Responses())
|
|
772
|
+
metergraph.wrap(client, provider="openai")
|
|
773
|
+
manual_trace_id = "a" * 32
|
|
774
|
+
parent_span_id = "b" * 16
|
|
775
|
+
|
|
776
|
+
with metergraph.trace(
|
|
777
|
+
"checkout", trace_id=manual_trace_id, parent_span_id=parent_span_id
|
|
778
|
+
):
|
|
779
|
+
client.responses.create(model="test", input="first")
|
|
780
|
+
with metergraph.trace("nested-reuses-active"):
|
|
781
|
+
client.responses.create(model="test", input="second")
|
|
782
|
+
with metergraph.trace("explicit-fork", trace_id="c" * 32):
|
|
783
|
+
client.responses.create(model="test", input="forked")
|
|
784
|
+
|
|
785
|
+
@metergraph.trace("async-checkout")
|
|
786
|
+
async def traced_async():
|
|
787
|
+
await asyncio.sleep(0)
|
|
788
|
+
client.responses.create(model="test", input="third")
|
|
789
|
+
client.responses.create(model="test", input="fourth")
|
|
790
|
+
|
|
791
|
+
asyncio.run(traced_async())
|
|
792
|
+
|
|
793
|
+
assert {row["trace_id"] for row in rows.rows[:2]} == {manual_trace_id}
|
|
794
|
+
assert {row["trace_name"] for row in rows.rows[:2]} == {"checkout"}
|
|
795
|
+
assert {row["parent_span_id"] for row in rows.rows[:2]} == {parent_span_id}
|
|
796
|
+
assert len({row["span_id"] for row in rows.rows[:2]}) == 2
|
|
797
|
+
assert rows.rows[2]["trace_id"] == "c" * 32
|
|
798
|
+
assert rows.rows[2]["trace_name"] == "explicit-fork"
|
|
799
|
+
assert rows.rows[3]["trace_id"] == rows.rows[4]["trace_id"]
|
|
800
|
+
assert rows.rows[3]["trace_name"] == "async-checkout"
|
|
801
|
+
assert rows.rows[3]["trace_id"] != manual_trace_id
|
|
802
|
+
_capture.set_runtime(None)
|
|
803
|
+
|
|
804
|
+
|
|
805
|
+
def test_trace_capture_override_scrubbing_redaction_and_utf8_limits(tmp_path):
|
|
806
|
+
rows = Rows()
|
|
807
|
+
|
|
808
|
+
def redact(value, kind):
|
|
809
|
+
return value.replace("customer-secret", f"<redacted-{kind}>")
|
|
810
|
+
|
|
811
|
+
runtime = Runtime(
|
|
812
|
+
rows,
|
|
813
|
+
Options(
|
|
814
|
+
app_root=str(tmp_path),
|
|
815
|
+
capture_text=True,
|
|
816
|
+
redact=redact,
|
|
817
|
+
text_max_bytes=100 * 1024,
|
|
818
|
+
),
|
|
819
|
+
)
|
|
820
|
+
hidden = runtime.call_state(
|
|
821
|
+
"openai", "responses", {"model": "test", "input": "hidden"}
|
|
822
|
+
)
|
|
823
|
+
with metergraph.trace("sensitive", capture_text=False):
|
|
824
|
+
# CaptureContext is read when the call starts.
|
|
825
|
+
hidden = runtime.call_state(
|
|
826
|
+
"openai", "responses", {"model": "test", "input": "hidden"}
|
|
827
|
+
)
|
|
828
|
+
hidden.finish(response("hidden output"))
|
|
829
|
+
assert rows.rows[0]["content_opted_in"] is False
|
|
830
|
+
assert rows.rows[0]["request_json"] is None
|
|
831
|
+
assert rows.rows[0]["response_text"] is None
|
|
832
|
+
|
|
833
|
+
call = runtime.call_state(
|
|
834
|
+
"openai",
|
|
835
|
+
"responses",
|
|
836
|
+
{
|
|
837
|
+
"model": "test",
|
|
838
|
+
"authorization": "Bearer provider-secret",
|
|
839
|
+
"headers": {"x-api-key": "provider-secret"},
|
|
840
|
+
"input": "customer-secret" + ("ü" * 80_000),
|
|
841
|
+
},
|
|
842
|
+
)
|
|
843
|
+
call.finish(response("customer-secret" + ("é" * 80_000)))
|
|
844
|
+
row = rows.rows[1]
|
|
845
|
+
assert "provider-secret" not in row["request_json"]
|
|
846
|
+
assert "customer-secret" not in row["request_json"]
|
|
847
|
+
assert "customer-secret" not in row["response_text"]
|
|
848
|
+
assert len(row["request_json"].encode()) <= 100 * 1024
|
|
849
|
+
assert len(row["response_text"].encode()) <= 100 * 1024
|
|
850
|
+
assert row["text_truncated"] is True
|
|
851
|
+
assert row["request_json"].endswith("<metergraph:truncated>")
|
|
852
|
+
assert row["response_text"].endswith("<metergraph:truncated>")
|
|
853
|
+
|
|
854
|
+
|
|
709
855
|
def test_wrap_async_errors_are_recorded_and_original_error_is_raised(tmp_path):
|
|
710
856
|
rows = Rows()
|
|
711
857
|
_capture.set_runtime(
|
|
@@ -730,6 +876,10 @@ def test_wrap_async_errors_are_recorded_and_original_error_is_raised(tmp_path):
|
|
|
730
876
|
asyncio.run(run())
|
|
731
877
|
assert rows.rows[0]["error"] is True
|
|
732
878
|
assert rows.rows[0]["error_type"] == "ValueError"
|
|
879
|
+
assert captured_response(rows.rows[0])["error"] == {
|
|
880
|
+
"type": "ValueError",
|
|
881
|
+
"message": "provider down",
|
|
882
|
+
}
|
|
733
883
|
_capture.set_runtime(None)
|
|
734
884
|
|
|
735
885
|
|
|
@@ -747,6 +897,44 @@ def test_stream_records_ttft_and_final_usage(tmp_path):
|
|
|
747
897
|
choices=[SimpleNamespace(delta=SimpleNamespace(content="hi"))],
|
|
748
898
|
usage=None,
|
|
749
899
|
),
|
|
900
|
+
SimpleNamespace(
|
|
901
|
+
choices=[
|
|
902
|
+
SimpleNamespace(
|
|
903
|
+
delta=SimpleNamespace(
|
|
904
|
+
content=None,
|
|
905
|
+
tool_calls=[
|
|
906
|
+
SimpleNamespace(
|
|
907
|
+
index=0,
|
|
908
|
+
id="call_1",
|
|
909
|
+
function=SimpleNamespace(
|
|
910
|
+
name="lookup", arguments='{"id":'
|
|
911
|
+
),
|
|
912
|
+
)
|
|
913
|
+
],
|
|
914
|
+
)
|
|
915
|
+
)
|
|
916
|
+
],
|
|
917
|
+
usage=None,
|
|
918
|
+
),
|
|
919
|
+
SimpleNamespace(
|
|
920
|
+
choices=[
|
|
921
|
+
SimpleNamespace(
|
|
922
|
+
delta=SimpleNamespace(
|
|
923
|
+
content=None,
|
|
924
|
+
tool_calls=[
|
|
925
|
+
SimpleNamespace(
|
|
926
|
+
index=0,
|
|
927
|
+
id=None,
|
|
928
|
+
function=SimpleNamespace(
|
|
929
|
+
name=None, arguments='"ord_1"}'
|
|
930
|
+
),
|
|
931
|
+
)
|
|
932
|
+
],
|
|
933
|
+
)
|
|
934
|
+
)
|
|
935
|
+
],
|
|
936
|
+
usage=None,
|
|
937
|
+
),
|
|
750
938
|
SimpleNamespace(
|
|
751
939
|
choices=[],
|
|
752
940
|
usage=SimpleNamespace(
|
|
@@ -767,14 +955,22 @@ def test_stream_records_ttft_and_final_usage(tmp_path):
|
|
|
767
955
|
)
|
|
768
956
|
# The SDK-added OpenAI usage-only chunk is consumed for metering but is
|
|
769
957
|
# not exposed to an application that did not ask for it.
|
|
770
|
-
assert len(chunks) ==
|
|
958
|
+
assert len(chunks) == 3
|
|
771
959
|
assert rows.rows[0]["stream"] is True
|
|
772
960
|
assert rows.rows[0]["ttft_ms"] is not None
|
|
773
961
|
assert rows.rows[0]["input_tokens"] == 2
|
|
774
962
|
assert rows.rows[0]["cache_read_tokens"] == 1
|
|
775
963
|
assert rows.rows[0]["cache_write_tokens"] == 2
|
|
776
964
|
assert "cost_usd" not in rows.rows[0]
|
|
777
|
-
assert rows.rows[0]["
|
|
965
|
+
assert captured_response(rows.rows[0])["content"] == "hi"
|
|
966
|
+
assert captured_response(rows.rows[0])["tool_calls"][0] == {
|
|
967
|
+
"call_id": "call_1",
|
|
968
|
+
"name": "lookup",
|
|
969
|
+
"arguments": {"id": "ord_1"},
|
|
970
|
+
"result": None,
|
|
971
|
+
"status": "requested",
|
|
972
|
+
"idempotency": "non_idempotent",
|
|
973
|
+
}
|
|
778
974
|
assert rows.rows[0]["request_json"].find("include_usage") >= 0
|
|
779
975
|
_capture.set_runtime(None)
|
|
780
976
|
|
|
@@ -829,7 +1025,7 @@ def test_async_stream_awaits_anthropic_final_message(tmp_path):
|
|
|
829
1025
|
assert rows.rows[0]["cache_write_5m_tokens"] == 2
|
|
830
1026
|
assert rows.rows[0]["cache_write_1h_tokens"] == 3
|
|
831
1027
|
assert "cost_usd" not in rows.rows[0]
|
|
832
|
-
assert rows.rows[0]["
|
|
1028
|
+
assert captured_response(rows.rows[0])["content"] == "ok"
|
|
833
1029
|
_capture.set_runtime(None)
|
|
834
1030
|
|
|
835
1031
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|