metergraph 0.2.1__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (21) hide show
  1. {metergraph-0.2.1 → metergraph-0.3.0}/PKG-INFO +32 -12
  2. {metergraph-0.2.1 → metergraph-0.3.0}/README.md +31 -11
  3. {metergraph-0.2.1 → metergraph-0.3.0}/pyproject.toml +1 -1
  4. {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/__init__.py +14 -3
  5. {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/_capture.py +195 -14
  6. {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/_context.py +89 -0
  7. {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/_template.py +18 -2
  8. {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/_version.py +1 -1
  9. {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph.egg-info/PKG-INFO +32 -12
  10. {metergraph-0.2.1 → metergraph-0.3.0}/tests/test_sdk.py +210 -14
  11. {metergraph-0.2.1 → metergraph-0.3.0}/setup.cfg +0 -0
  12. {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/_config.py +0 -0
  13. {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/_failure_log.py +0 -0
  14. {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/_track.py +0 -0
  15. {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph/_transport.py +0 -0
  16. {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph.egg-info/SOURCES.txt +0 -0
  17. {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph.egg-info/dependency_links.txt +0 -0
  18. {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph.egg-info/requires.txt +0 -0
  19. {metergraph-0.2.1 → metergraph-0.3.0}/src/metergraph.egg-info/top_level.txt +0 -0
  20. {metergraph-0.2.1 → metergraph-0.3.0}/tests/test_real_client_integration.py +0 -0
  21. {metergraph-0.2.1 → metergraph-0.3.0}/tests/test_seam_reality.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: metergraph
3
- Version: 0.2.1
3
+ Version: 0.3.0
4
4
  Summary: Fire-and-forget LLM spend capture for Metergraph
5
5
  Author: Pioneer Square Labs
6
6
  License-Expression: Apache-2.0
@@ -34,9 +34,10 @@ from openai import OpenAI
34
34
  client = metergraph.wrap(OpenAI())
35
35
  metergraph.set_session("ticket-123")
36
36
 
37
- with metergraph.route("ticket-classifier", unit="answer", capture_text=True):
38
- model = metergraph.model_for("ticket-classifier", default="gpt-4.1-mini")
39
- client.chat.completions.create(model=model, messages=[...])
37
+ with metergraph.trace("ticket-workflow"):
38
+ with metergraph.route("ticket-classifier", unit="answer"):
39
+ model = metergraph.model_for("ticket-classifier", default="gpt-4.1-mini")
40
+ client.chat.completions.create(model=model, messages=[...])
40
41
 
41
42
  # Emit this after the user-visible task resolves. It shares the bounded async
42
43
  # transport and contains no prompt or output content.
@@ -54,15 +55,30 @@ Configuration:
54
55
 
55
56
  - `METERGRAPH_APP_TOKEN` — required bearer token
56
57
  - `METERGRAPH_INGEST_URL` — optional override; defaults to the hosted HTTPS endpoint
57
- - `METERGRAPH_CAPTURE_TEXT=1` — opt in to content capture globally; default is metadata-only
58
+ - `METERGRAPH_CAPTURE_TEXT=0` — opt out of content capture globally
58
59
  - `METERGRAPH_DISABLED=1` — process kill switch
59
60
  - `METERGRAPH_QUEUE_SIZE`, `METERGRAPH_BATCH_SIZE`, `METERGRAPH_FLUSH_SECONDS`
60
61
 
61
62
  Delivery is bounded and off the request path. Queue overflow or a collector
62
63
  outage drops capture and increments internal counters; it never changes the
63
64
  provider call. Each wire batch is bounded to 512 KiB after optional gzip.
64
- `capture_text=True` or `False` on `route()` overrides the global content policy
65
- for that route. Config reads are ETag-aware and fail open to the default model.
65
+ SDK 0.3 captures the scrubbed provider request and a normalized response
66
+ envelope, including assistant content and tool calls, by default. Provider
67
+ credentials and transport headers are removed. Request and response are each
68
+ limited to 100 KiB of UTF-8 with an explicit truncation marker.
69
+ `capture_text=False` on `route()` or `trace()` overrides the global content
70
+ policy for a sensitive operation. The equivalent initialization option is
71
+ `metergraph.init(capture_text=False)`. The public open-source server continues
72
+ to discard content even when the SDK sends it; the hosted dashboard retains
73
+ content under the workspace retention period.
74
+
75
+ `metergraph.trace(name, trace_id=..., parent_span_id=...)` is a sync/async
76
+ context manager and decorator. Calls inside one trace share a trace ID and
77
+ receive distinct span IDs. Calls outside a trace become deterministic
78
+ single-span traces after ingestion. Manual IDs can join work across process
79
+ boundaries; automatic W3C HTTP propagation is not included.
80
+
81
+ Config reads are ETag-aware and fail open to the default model.
66
82
  `record_outcome` requires a stable session ID and the model actually used so a
67
83
  session-sticky canary can compare task completion and optional feedback,
68
84
  turn-count, escalation, abandonment, edit-distance, and regeneration signals.
@@ -88,11 +104,15 @@ genai.Client() construction in place, e.g. client = metergraph.wrap(OpenAI()).
88
104
  wrap() returns the same client and initializes itself from the environment:
89
105
  METERGRAPH_APP_TOKEN is required (capture is silently off without it) and
90
106
  METERGRAPH_INGEST_URL is only for self-hosted servers. Add both to
91
- .env.example, and never commit a real token. Capture is metadata-only (tokens,
92
- latency, model, no prompt/completion content) and fail-open, so do not change
93
- call sites, arguments, or error handling; sync, async, and streaming work
94
- unchanged. Attribution to the calling function is automatic; optionally pin
95
- stable names on key LLM-calling functions with @metergraph.track. On
107
+ .env.example, and never commit a real token. SDK 0.3 captures scrubbed provider
108
+ requests and normalized responses by default for the hosted dashboard; use
109
+ METERGRAPH_CAPTURE_TEXT=0 or capture_text=False around sensitive operations.
110
+ Provider credentials and transport headers must never be captured. Capture is
111
+ fail-open, so do not change call sites, arguments, or error handling; sync,
112
+ async, and streaming work unchanged. Use metergraph.trace("stable-name") to
113
+ group multi-call workflows. Attribution to the calling function is automatic;
114
+ optionally pin stable names on key LLM-calling functions with
115
+ @metergraph.track. On
96
116
  serverless, call metergraph.flush() before the handler returns. When done,
97
117
  list every client you wrapped and flag LLM calls made outside the official
98
118
  openai / anthropic / google-genai SDKs, since those are not captured.
@@ -13,9 +13,10 @@ from openai import OpenAI
13
13
  client = metergraph.wrap(OpenAI())
14
14
  metergraph.set_session("ticket-123")
15
15
 
16
- with metergraph.route("ticket-classifier", unit="answer", capture_text=True):
17
- model = metergraph.model_for("ticket-classifier", default="gpt-4.1-mini")
18
- client.chat.completions.create(model=model, messages=[...])
16
+ with metergraph.trace("ticket-workflow"):
17
+ with metergraph.route("ticket-classifier", unit="answer"):
18
+ model = metergraph.model_for("ticket-classifier", default="gpt-4.1-mini")
19
+ client.chat.completions.create(model=model, messages=[...])
19
20
 
20
21
  # Emit this after the user-visible task resolves. It shares the bounded async
21
22
  # transport and contains no prompt or output content.
@@ -33,15 +34,30 @@ Configuration:
33
34
 
34
35
  - `METERGRAPH_APP_TOKEN` — required bearer token
35
36
  - `METERGRAPH_INGEST_URL` — optional override; defaults to the hosted HTTPS endpoint
36
- - `METERGRAPH_CAPTURE_TEXT=1` — opt in to content capture globally; default is metadata-only
37
+ - `METERGRAPH_CAPTURE_TEXT=0` — opt out of content capture globally
37
38
  - `METERGRAPH_DISABLED=1` — process kill switch
38
39
  - `METERGRAPH_QUEUE_SIZE`, `METERGRAPH_BATCH_SIZE`, `METERGRAPH_FLUSH_SECONDS`
39
40
 
40
41
  Delivery is bounded and off the request path. Queue overflow or a collector
41
42
  outage drops capture and increments internal counters; it never changes the
42
43
  provider call. Each wire batch is bounded to 512 KiB after optional gzip.
43
- `capture_text=True` or `False` on `route()` overrides the global content policy
44
- for that route. Config reads are ETag-aware and fail open to the default model.
44
+ SDK 0.3 captures the scrubbed provider request and a normalized response
45
+ envelope, including assistant content and tool calls, by default. Provider
46
+ credentials and transport headers are removed. Request and response are each
47
+ limited to 100 KiB of UTF-8 with an explicit truncation marker.
48
+ `capture_text=False` on `route()` or `trace()` overrides the global content
49
+ policy for a sensitive operation. The equivalent initialization option is
50
+ `metergraph.init(capture_text=False)`. The public open-source server continues
51
+ to discard content even when the SDK sends it; the hosted dashboard retains
52
+ content under the workspace retention period.
53
+
54
+ `metergraph.trace(name, trace_id=..., parent_span_id=...)` is a sync/async
55
+ context manager and decorator. Calls inside one trace share a trace ID and
56
+ receive distinct span IDs. Calls outside a trace become deterministic
57
+ single-span traces after ingestion. Manual IDs can join work across process
58
+ boundaries; automatic W3C HTTP propagation is not included.
59
+
60
+ Config reads are ETag-aware and fail open to the default model.
45
61
  `record_outcome` requires a stable session ID and the model actually used so a
46
62
  session-sticky canary can compare task completion and optional feedback,
47
63
  turn-count, escalation, abandonment, edit-distance, and regeneration signals.
@@ -67,11 +83,15 @@ genai.Client() construction in place, e.g. client = metergraph.wrap(OpenAI()).
67
83
  wrap() returns the same client and initializes itself from the environment:
68
84
  METERGRAPH_APP_TOKEN is required (capture is silently off without it) and
69
85
  METERGRAPH_INGEST_URL is only for self-hosted servers. Add both to
70
- .env.example, and never commit a real token. Capture is metadata-only (tokens,
71
- latency, model, no prompt/completion content) and fail-open, so do not change
72
- call sites, arguments, or error handling; sync, async, and streaming work
73
- unchanged. Attribution to the calling function is automatic; optionally pin
74
- stable names on key LLM-calling functions with @metergraph.track. On
86
+ .env.example, and never commit a real token. SDK 0.3 captures scrubbed provider
87
+ requests and normalized responses by default for the hosted dashboard; use
88
+ METERGRAPH_CAPTURE_TEXT=0 or capture_text=False around sensitive operations.
89
+ Provider credentials and transport headers must never be captured. Capture is
90
+ fail-open, so do not change call sites, arguments, or error handling; sync,
91
+ async, and streaming work unchanged. Use metergraph.trace("stable-name") to
92
+ group multi-call workflows. Attribution to the calling function is automatic;
93
+ optionally pin stable names on key LLM-calling functions with
94
+ @metergraph.track. On
75
95
  serverless, call metergraph.flush() before the handler returns. When done,
76
96
  list every client you wrapped and flag LLM calls made outside the official
77
97
  openai / anthropic / google-genai SDKs, since those are not captured.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "metergraph"
3
- version = "0.2.1"
3
+ version = "0.3.0"
4
4
  description = "Fire-and-forget LLM spend capture for Metergraph"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"
@@ -12,7 +12,7 @@ from typing import Any, Callable
12
12
  from ._capture import Options, Runtime, set_runtime
13
13
  from ._capture import wrap as _wrap
14
14
  from ._config import ConfigPoller
15
- from ._context import route, set_session, set_tags, snapshot, wrap_executor
15
+ from ._context import route, set_session, set_tags, snapshot, trace, wrap_executor
16
16
  from ._track import track
17
17
  from ._transport import Writer
18
18
  from ._version import SDK_VERSION
@@ -73,7 +73,7 @@ def init(
73
73
  )
74
74
  options = Options(
75
75
  capture_text=(
76
- _env_bool("METERGRAPH_CAPTURE_TEXT", False)
76
+ _env_bool("METERGRAPH_CAPTURE_TEXT", True)
77
77
  if capture_text is None
78
78
  else capture_text
79
79
  ),
@@ -81,7 +81,17 @@ def init(
81
81
  app_root=os.path.realpath(app_root or os.getcwd()),
82
82
  skip_frames=tuple(skip_frames or ()),
83
83
  environment=environment or os.getenv("METERGRAPH_ENV"),
84
- text_max_bytes=int(os.getenv("METERGRAPH_TEXT_MAX_BYTES", "100000")),
84
+ text_max_bytes=min(
85
+ 100 * 1024,
86
+ max(
87
+ 1,
88
+ int(
89
+ os.getenv(
90
+ "METERGRAPH_TEXT_MAX_BYTES", str(100 * 1024)
91
+ )
92
+ ),
93
+ ),
94
+ ),
85
95
  )
86
96
  set_runtime(Runtime(_writer, options))
87
97
  _config = ConfigPoller(
@@ -214,6 +224,7 @@ __all__ = [
214
224
  "set_tags",
215
225
  "shutdown",
216
226
  "track",
227
+ "trace",
217
228
  "wrap",
218
229
  "wrap_executor",
219
230
  ]
@@ -8,6 +8,7 @@ import json
8
8
  import logging
9
9
  import os
10
10
  import platform
11
+ import secrets
11
12
  import sys
12
13
  import time
13
14
  from dataclasses import dataclass
@@ -223,11 +224,68 @@ def _request_id(response: Any) -> str | None:
223
224
  value = (
224
225
  _get(response, "_request_id")
225
226
  or _get(response, "response_id")
227
+ or _get(response, "responseId")
226
228
  or _get(response, "id")
227
229
  )
228
230
  return str(value) if value is not None else None
229
231
 
230
232
 
233
+ def _response_content(response: Any, aggregate_text: str | None = None) -> Any:
234
+ if aggregate_text is not None:
235
+ return aggregate_text
236
+ direct = _get(response, "output_text") or _get(response, "text")
237
+ if direct is not None:
238
+ return scrub(direct)
239
+ choice = _first(_get(response, "choices"))
240
+ message = _get(choice, "message")
241
+ content = _get(message, "content")
242
+ if content is not None:
243
+ return scrub(content)
244
+ parsed = _get(message, "parsed")
245
+ if parsed is not None:
246
+ return scrub(parsed)
247
+ normalized_text = _response_text(response)
248
+ if normalized_text is not None:
249
+ return normalized_text
250
+ blocks = _get(response, "content")
251
+ if blocks is not None:
252
+ return scrub(blocks)
253
+ outputs = _get(response, "output")
254
+ if outputs is not None:
255
+ return scrub(outputs)
256
+ candidates = _get(response, "candidates")
257
+ if candidates is not None:
258
+ return scrub(candidates)
259
+ return None
260
+
261
+
262
+ def _response_envelope(
263
+ response: Any,
264
+ *,
265
+ aggregate_text: str | None,
266
+ tool_calls: list[dict] | None,
267
+ error: BaseException | None,
268
+ status: str,
269
+ ) -> dict[str, Any]:
270
+ envelope: dict[str, Any] = {
271
+ "role": "assistant",
272
+ "content": _response_content(response, aggregate_text),
273
+ "tool_calls": tool_calls or [],
274
+ "finish_reason": _stop_reason(response),
275
+ "request_id": _request_id(response),
276
+ "model": _get(response, "model")
277
+ or _get(response, "model_version")
278
+ or _get(response, "modelVersion"),
279
+ "status": status,
280
+ }
281
+ if error is not None:
282
+ envelope["error"] = {
283
+ "type": type(error).__name__,
284
+ "message": str(error),
285
+ }
286
+ return {key: value for key, value in envelope.items() if value is not None}
287
+
288
+
231
289
  def _tool_names(request: Mapping[str, Any]) -> list[dict[str, str]] | None:
232
290
  tools = request.get("tools")
233
291
  if not isinstance(tools, list):
@@ -273,7 +331,11 @@ def _tool_policies(request: Mapping[str, Any]) -> dict[str, str]:
273
331
  return policies
274
332
 
275
333
 
276
- def _tool_events(request: Mapping[str, Any], response: Any) -> list[dict] | None:
334
+ def _tool_events(
335
+ request: Mapping[str, Any],
336
+ response: Any,
337
+ stream_chunks: list[Any] | None = None,
338
+ ) -> list[dict] | None:
277
339
  """Normalize completed history and newly requested provider tool calls."""
278
340
  policies = _tool_policies(request)
279
341
  calls: dict[str, dict] = {}
@@ -320,9 +382,37 @@ def _tool_events(request: Mapping[str, Any], response: Any) -> list[dict] | None
320
382
  bool(_get(block, "is_error", False)),
321
383
  )
322
384
 
385
+ def gemini_parts(value: Any) -> None:
386
+ parts = _get(value, "parts")
387
+ if not isinstance(parts, list):
388
+ return
389
+ for part in parts:
390
+ function_call = _get(part, "function_call") or _get(
391
+ part, "functionCall"
392
+ )
393
+ if function_call is not None:
394
+ call(
395
+ _get(function_call, "id"),
396
+ _get(function_call, "name"),
397
+ _get(function_call, "args")
398
+ or _get(function_call, "arguments"),
399
+ )
400
+ function_response = _get(part, "function_response") or _get(
401
+ part, "functionResponse"
402
+ )
403
+ if function_response is not None:
404
+ complete(
405
+ _get(function_response, "id")
406
+ or _get(function_response, "name"),
407
+ _get(function_response, "response"),
408
+ False,
409
+ )
410
+
323
411
  history = request.get("messages")
324
412
  if not isinstance(history, list):
325
413
  history = request.get("input")
414
+ if not isinstance(history, list):
415
+ history = request.get("contents")
326
416
  if isinstance(history, list):
327
417
  for message in history:
328
418
  for tool_call in _get(message, "tool_calls", []) or []:
@@ -353,6 +443,7 @@ def _tool_events(request: Mapping[str, Any], response: Any) -> list[dict] | None
353
443
  bool(_get(message, "is_error", False)),
354
444
  )
355
445
  content_blocks(_get(message, "content"))
446
+ gemini_parts(message)
356
447
 
357
448
  choice = _first(_get(response, "choices"))
358
449
  response_message = _get(choice, "message")
@@ -372,6 +463,60 @@ def _tool_events(request: Mapping[str, Any], response: Any) -> list[dict] | None
372
463
  _get(output, "name"),
373
464
  _get(output, "arguments"),
374
465
  )
466
+ for candidate in _get(response, "candidates", []) or []:
467
+ gemini_parts(_get(candidate, "content"))
468
+
469
+ openai_deltas: dict[str, dict[str, str]] = {}
470
+ anthropic_deltas: dict[str, dict[str, str]] = {}
471
+ for chunk in stream_chunks or []:
472
+ for choice in _get(chunk, "choices", []) or []:
473
+ delta = _get(choice, "delta")
474
+ for position, tool_call in enumerate(
475
+ _get(delta, "tool_calls", []) or []
476
+ ):
477
+ key = str(
478
+ _get(tool_call, "index")
479
+ if _get(tool_call, "index") is not None
480
+ else position
481
+ )
482
+ item = openai_deltas.setdefault(
483
+ key, {"id": key, "name": "", "arguments": ""}
484
+ )
485
+ if _get(tool_call, "id"):
486
+ item["id"] = str(_get(tool_call, "id"))
487
+ fn = _get(tool_call, "function")
488
+ if _get(fn, "name"):
489
+ item["name"] += str(_get(fn, "name"))
490
+ if _get(fn, "arguments"):
491
+ item["arguments"] += str(_get(fn, "arguments"))
492
+ kind = _get(chunk, "type")
493
+ if kind == "content_block_start":
494
+ block = _get(chunk, "content_block")
495
+ if _get(block, "type") == "tool_use":
496
+ key = str(_get(chunk, "index", len(anthropic_deltas)))
497
+ initial_input = scrub(_get(block, "input", {}))
498
+ anthropic_deltas[key] = {
499
+ "id": str(_get(block, "id") or key),
500
+ "name": str(_get(block, "name") or ""),
501
+ "arguments": (
502
+ ""
503
+ if initial_input == {}
504
+ else json.dumps(initial_input, separators=(",", ":"))
505
+ ),
506
+ }
507
+ elif kind == "content_block_delta":
508
+ delta = _get(chunk, "delta")
509
+ if _get(delta, "type") == "input_json_delta":
510
+ key = str(_get(chunk, "index", "0"))
511
+ item = anthropic_deltas.setdefault(
512
+ key, {"id": key, "name": "", "arguments": ""}
513
+ )
514
+ item["arguments"] += str(_get(delta, "partial_json") or "")
515
+ for candidate in _get(chunk, "candidates", []) or []:
516
+ gemini_parts(_get(candidate, "content"))
517
+ for item in [*openai_deltas.values(), *anthropic_deltas.values()]:
518
+ if item["name"]:
519
+ call(item["id"], item["name"], item["arguments"])
375
520
 
376
521
  return [calls[key] for key in order] or None
377
522
 
@@ -399,12 +544,12 @@ def _capture_frames(
399
544
 
400
545
  @dataclass
401
546
  class Options:
402
- capture_text: bool = False
547
+ capture_text: bool = True
403
548
  redact: Callable[[str, str], str] | None = None
404
549
  app_root: str = os.getcwd()
405
550
  skip_frames: tuple[str, ...] = ()
406
551
  environment: str | None = None
407
- text_max_bytes: int = 100_000
552
+ text_max_bytes: int = 100 * 1024
408
553
 
409
554
 
410
555
  class Runtime:
@@ -442,6 +587,14 @@ class Runtime:
442
587
  func=context.func_name or func,
443
588
  module=context.func_module or module,
444
589
  frames=frames,
590
+ trace_id=context.trace_id or secrets.token_hex(16),
591
+ span_id=secrets.token_hex(8),
592
+ parent_span_id=context.parent_span_id,
593
+ trace_name=context.trace_name
594
+ or context.route
595
+ or context.func_name
596
+ or func
597
+ or endpoint,
445
598
  )
446
599
 
447
600
  def _text(
@@ -476,6 +629,10 @@ class CallState:
476
629
  func: str | None
477
630
  module: str | None
478
631
  frames: list[dict]
632
+ trace_id: str
633
+ span_id: str
634
+ parent_span_id: str | None
635
+ trace_name: str
479
636
  done: bool = False
480
637
 
481
638
  def finish(
@@ -487,6 +644,7 @@ class CallState:
487
644
  stream: bool = False,
488
645
  ttft_ms: int | None = None,
489
646
  response_text: str | None = None,
647
+ stream_chunks: list[Any] | None = None,
490
648
  ) -> None:
491
649
  if self.done:
492
650
  return
@@ -502,17 +660,13 @@ class CallState:
502
660
  "request",
503
661
  enabled=capture_text,
504
662
  )
505
- response_json, response_truncated = self.runtime._text(
506
- response_text if response_text is not None else _response_text(response),
507
- "response",
508
- enabled=capture_text,
509
- )
510
- tool_calls = _tool_events(request_clean, response)
663
+ full_tool_calls = _tool_events(request_clean, response, stream_chunks)
664
+ tool_calls = full_tool_calls
511
665
  tool_truncated = False
512
666
  if tool_calls and capture_text:
513
667
  encoded_tools, tool_truncated = self.runtime._text(
514
668
  json.dumps(tool_calls, separators=(",", ":"), default=repr),
515
- "tool_calls",
669
+ "response",
516
670
  enabled=True,
517
671
  )
518
672
  try:
@@ -529,6 +683,25 @@ class CallState:
529
683
  }
530
684
  for item in tool_calls
531
685
  ]
686
+ effective_status = status or (
687
+ "error" if error else _stop_reason(response) or "success"
688
+ )
689
+ response_json, response_truncated = self.runtime._text(
690
+ json.dumps(
691
+ _response_envelope(
692
+ response,
693
+ aggregate_text=response_text,
694
+ tool_calls=full_tool_calls if capture_text else None,
695
+ error=error,
696
+ status=effective_status,
697
+ ),
698
+ ensure_ascii=False,
699
+ separators=(",", ":"),
700
+ default=repr,
701
+ ),
702
+ "response",
703
+ enabled=capture_text,
704
+ )
532
705
  row: dict[str, Any] = {
533
706
  "ts": self.ts,
534
707
  "route": self.context.route,
@@ -536,10 +709,13 @@ class CallState:
536
709
  "model": self.request.get("model"),
537
710
  **_usage(response),
538
711
  "latency_ms": round((time.perf_counter() - self.started) * 1000),
539
- "status": status
540
- or ("error" if error else _stop_reason(response) or "success"),
712
+ "status": effective_status,
541
713
  "session_id": self.context.session_id,
542
714
  "conversation_id": self.context.session_id,
715
+ "trace_id": self.trace_id,
716
+ "span_id": self.span_id,
717
+ "parent_span_id": self.parent_span_id,
718
+ "trace_name": self.trace_name,
543
719
  "template_hash": template_hash(self.request),
544
720
  "unit_name": self.context.unit_name,
545
721
  "unit_count": self.context.unit_count,
@@ -548,8 +724,8 @@ class CallState:
548
724
  "request_id": _request_id(response),
549
725
  "batch": self.request.get("batch") is True,
550
726
  "batch_custom_id": self.request.get("batch_custom_id"),
551
- # The worker requires an explicit positive stamp before sending
552
- # any content to Bedrock. Missing/old rows therefore fail closed.
727
+ # An explicit false is the application-level sensitive-operation
728
+ # opt-out. Hosted ingestion otherwise preserves available content.
553
729
  "content_opted_in": capture_text,
554
730
  "request_json": request_json,
555
731
  "response_text": response_json,
@@ -580,10 +756,12 @@ class _StreamState:
580
756
  self.iterator = None
581
757
  self.last = None
582
758
  self.parts: list[str] = []
759
+ self.chunks: list[Any] = []
583
760
  self.ttft_ms: int | None = None
584
761
 
585
762
  def chunk(self, value: Any) -> Any:
586
763
  self.last = value
764
+ self.chunks.append(value)
587
765
  text = _chunk_text(value)
588
766
  if text:
589
767
  if self.ttft_ms is None:
@@ -614,6 +792,7 @@ class _StreamState:
614
792
  stream=True,
615
793
  ttft_ms=self.ttft_ms,
616
794
  response_text="".join(self.parts) or None,
795
+ stream_chunks=self.chunks,
617
796
  )
618
797
 
619
798
  async def finish_async(
@@ -636,6 +815,7 @@ class _StreamState:
636
815
  stream=True,
637
816
  ttft_ms=self.ttft_ms,
638
817
  response_text="".join(self.parts) or None,
818
+ stream_chunks=self.chunks,
639
819
  )
640
820
 
641
821
 
@@ -754,6 +934,7 @@ class AsyncStream:
754
934
  stream=True,
755
935
  ttft_ms=self._state.ttft_ms,
756
936
  response_text="".join(self._state.parts) or None,
937
+ stream_chunks=self._state.chunks,
757
938
  )
758
939
  except Exception:
759
940
  pass
@@ -5,6 +5,7 @@ from __future__ import annotations
5
5
  import contextvars
6
6
  import functools
7
7
  import inspect
8
+ import secrets
8
9
  from concurrent.futures import Executor
9
10
  from dataclasses import dataclass, field, replace
10
11
  from typing import Any, Callable, Mapping
@@ -20,6 +21,9 @@ class CaptureContext:
20
21
  capture_text: bool | None = None
21
22
  func_name: str | None = None
22
23
  func_module: str | None = None
24
+ trace_id: str | None = None
25
+ trace_name: str | None = None
26
+ parent_span_id: str | None = None
23
27
 
24
28
 
25
29
  _current: contextvars.ContextVar[CaptureContext] = contextvars.ContextVar(
@@ -108,6 +112,91 @@ class route:
108
112
  return wrapped
109
113
 
110
114
 
115
+ class trace:
116
+ """Logical trace context manager and sync/async decorator."""
117
+
118
+ def __init__(
119
+ self,
120
+ name: str,
121
+ *,
122
+ trace_id: str | None = None,
123
+ parent_span_id: str | None = None,
124
+ capture_text: bool | None = None,
125
+ ) -> None:
126
+ self.name = str(name)
127
+ self.trace_id = str(trace_id).strip() if trace_id is not None else None
128
+ self.parent_span_id = (
129
+ str(parent_span_id).strip() if parent_span_id is not None else None
130
+ )
131
+ self.capture_text = (
132
+ bool(capture_text) if capture_text is not None else None
133
+ )
134
+ self._token: contextvars.Token[CaptureContext] | None = None
135
+
136
+ def __enter__(self) -> "trace":
137
+ current = snapshot()
138
+ requested = self.trace_id
139
+ reuse = current.trace_id is not None and (
140
+ requested is None or requested == current.trace_id
141
+ )
142
+ self._token = _current.set(
143
+ replace(
144
+ current,
145
+ trace_id=(
146
+ current.trace_id
147
+ if reuse
148
+ else requested or secrets.token_hex(16)
149
+ ),
150
+ trace_name=current.trace_name if reuse else self.name,
151
+ parent_span_id=(
152
+ self.parent_span_id
153
+ if self.parent_span_id is not None
154
+ else current.parent_span_id
155
+ if reuse
156
+ else None
157
+ ),
158
+ capture_text=(
159
+ self.capture_text
160
+ if self.capture_text is not None
161
+ else current.capture_text
162
+ ),
163
+ )
164
+ )
165
+ return self
166
+
167
+ def __exit__(self, exc_type, exc, tb) -> None:
168
+ if self._token is not None:
169
+ _current.reset(self._token)
170
+ self._token = None
171
+
172
+ def __call__(self, fn: Callable):
173
+ if inspect.iscoroutinefunction(fn):
174
+
175
+ @functools.wraps(fn)
176
+ async def async_wrapped(*args, **kwargs):
177
+ with type(self)(
178
+ self.name,
179
+ trace_id=self.trace_id,
180
+ parent_span_id=self.parent_span_id,
181
+ capture_text=self.capture_text,
182
+ ):
183
+ return await fn(*args, **kwargs)
184
+
185
+ return async_wrapped
186
+
187
+ @functools.wraps(fn)
188
+ def wrapped(*args, **kwargs):
189
+ with type(self)(
190
+ self.name,
191
+ trace_id=self.trace_id,
192
+ parent_span_id=self.parent_span_id,
193
+ capture_text=self.capture_text,
194
+ ):
195
+ return fn(*args, **kwargs)
196
+
197
+ return wrapped
198
+
199
+
111
200
  def set_session(session_id: str | None) -> None:
112
201
  _current.set(
113
202
  replace(snapshot(), session_id=str(session_id) if session_id else None)
@@ -14,7 +14,23 @@ _EMAIL = re.compile(r"\b[^\s@]+@[^\s@]+\.[^\s@]+\b")
14
14
  _URL = re.compile(r"\bhttps?://\S+")
15
15
  _NUMBER = re.compile(r"(?<![A-Za-z])[-+]?\d+(?:\.\d+)?(?![A-Za-z])")
16
16
  _LONG_TOKEN = re.compile(r"\b[A-Za-z0-9_-]{24,}\b")
17
- _SENSITIVE_KEYS = {"api_key", "apikey", "authorization", "headers", "token", "secret"}
17
+ _SENSITIVE_KEYS = {
18
+ "api-key",
19
+ "api_key",
20
+ "apikey",
21
+ "authorization",
22
+ "client_secret",
23
+ "cookie",
24
+ "headers",
25
+ "id_token",
26
+ "password",
27
+ "proxy-authorization",
28
+ "refresh_token",
29
+ "secret",
30
+ "set-cookie",
31
+ "token",
32
+ "x-api-key",
33
+ }
18
34
 
19
35
 
20
36
  def _normalize_text(value: str) -> str:
@@ -31,7 +47,7 @@ def scrub(value: Any) -> Any:
31
47
  return {
32
48
  str(k): scrub(v)
33
49
  for k, v in value.items()
34
- if str(k).lower() not in _SENSITIVE_KEYS
50
+ if str(k).strip().lower() not in _SENSITIVE_KEYS
35
51
  }
36
52
  if isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)):
37
53
  return [scrub(item) for item in value]
@@ -8,4 +8,4 @@ import importlib.metadata
8
8
  try:
9
9
  SDK_VERSION = importlib.metadata.version("metergraph")
10
10
  except Exception:
11
- SDK_VERSION = "0.2.1"
11
+ SDK_VERSION = "0.3.0"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: metergraph
3
- Version: 0.2.1
3
+ Version: 0.3.0
4
4
  Summary: Fire-and-forget LLM spend capture for Metergraph
5
5
  Author: Pioneer Square Labs
6
6
  License-Expression: Apache-2.0
@@ -34,9 +34,10 @@ from openai import OpenAI
34
34
  client = metergraph.wrap(OpenAI())
35
35
  metergraph.set_session("ticket-123")
36
36
 
37
- with metergraph.route("ticket-classifier", unit="answer", capture_text=True):
38
- model = metergraph.model_for("ticket-classifier", default="gpt-4.1-mini")
39
- client.chat.completions.create(model=model, messages=[...])
37
+ with metergraph.trace("ticket-workflow"):
38
+ with metergraph.route("ticket-classifier", unit="answer"):
39
+ model = metergraph.model_for("ticket-classifier", default="gpt-4.1-mini")
40
+ client.chat.completions.create(model=model, messages=[...])
40
41
 
41
42
  # Emit this after the user-visible task resolves. It shares the bounded async
42
43
  # transport and contains no prompt or output content.
@@ -54,15 +55,30 @@ Configuration:
54
55
 
55
56
  - `METERGRAPH_APP_TOKEN` — required bearer token
56
57
  - `METERGRAPH_INGEST_URL` — optional override; defaults to the hosted HTTPS endpoint
57
- - `METERGRAPH_CAPTURE_TEXT=1` — opt in to content capture globally; default is metadata-only
58
+ - `METERGRAPH_CAPTURE_TEXT=0` — opt out of content capture globally
58
59
  - `METERGRAPH_DISABLED=1` — process kill switch
59
60
  - `METERGRAPH_QUEUE_SIZE`, `METERGRAPH_BATCH_SIZE`, `METERGRAPH_FLUSH_SECONDS`
60
61
 
61
62
  Delivery is bounded and off the request path. Queue overflow or a collector
62
63
  outage drops capture and increments internal counters; it never changes the
63
64
  provider call. Each wire batch is bounded to 512 KiB after optional gzip.
64
- `capture_text=True` or `False` on `route()` overrides the global content policy
65
- for that route. Config reads are ETag-aware and fail open to the default model.
65
+ SDK 0.3 captures the scrubbed provider request and a normalized response
66
+ envelope, including assistant content and tool calls, by default. Provider
67
+ credentials and transport headers are removed. Request and response are each
68
+ limited to 100 KiB of UTF-8 with an explicit truncation marker.
69
+ `capture_text=False` on `route()` or `trace()` overrides the global content
70
+ policy for a sensitive operation. The equivalent initialization option is
71
+ `metergraph.init(capture_text=False)`. The public open-source server continues
72
+ to discard content even when the SDK sends it; the hosted dashboard retains
73
+ content under the workspace retention period.
74
+
75
+ `metergraph.trace(name, trace_id=..., parent_span_id=...)` is a sync/async
76
+ context manager and decorator. Calls inside one trace share a trace ID and
77
+ receive distinct span IDs. Calls outside a trace become deterministic
78
+ single-span traces after ingestion. Manual IDs can join work across process
79
+ boundaries; automatic W3C HTTP propagation is not included.
80
+
81
+ Config reads are ETag-aware and fail open to the default model.
66
82
  `record_outcome` requires a stable session ID and the model actually used so a
67
83
  session-sticky canary can compare task completion and optional feedback,
68
84
  turn-count, escalation, abandonment, edit-distance, and regeneration signals.
@@ -88,11 +104,15 @@ genai.Client() construction in place, e.g. client = metergraph.wrap(OpenAI()).
88
104
  wrap() returns the same client and initializes itself from the environment:
89
105
  METERGRAPH_APP_TOKEN is required (capture is silently off without it) and
90
106
  METERGRAPH_INGEST_URL is only for self-hosted servers. Add both to
91
- .env.example, and never commit a real token. Capture is metadata-only (tokens,
92
- latency, model, no prompt/completion content) and fail-open, so do not change
93
- call sites, arguments, or error handling; sync, async, and streaming work
94
- unchanged. Attribution to the calling function is automatic; optionally pin
95
- stable names on key LLM-calling functions with @metergraph.track. On
107
+ .env.example, and never commit a real token. SDK 0.3 captures scrubbed provider
108
+ requests and normalized responses by default for the hosted dashboard; use
109
+ METERGRAPH_CAPTURE_TEXT=0 or capture_text=False around sensitive operations.
110
+ Provider credentials and transport headers must never be captured. Capture is
111
+ fail-open, so do not change call sites, arguments, or error handling; sync,
112
+ async, and streaming work unchanged. Use metergraph.trace("stable-name") to
113
+ group multi-call workflows. Attribution to the calling function is automatic;
114
+ optionally pin stable names on key LLM-calling functions with
115
+ @metergraph.track. On
96
116
  serverless, call metergraph.flush() before the handler returns. When done,
97
117
  list every client you wrapped and flag LLM calls made outside the official
98
118
  openai / anthropic / google-genai SDKs, since those are not captured.
@@ -63,6 +63,10 @@ def response(text="done"):
63
63
  )
64
64
 
65
65
 
66
+ def captured_response(row):
67
+ return json.loads(row["response_text"])
68
+
69
+
66
70
  def test_wrap_auto_init_does_not_latch_before_a_token_is_available():
67
71
  class Completions:
68
72
  def create(self, **kwargs):
@@ -283,7 +287,7 @@ def test_wrap_google_records_usage_and_endpoint(tmp_path):
283
287
  assert row["output_tokens"] == 20
284
288
  assert row["cache_read_tokens"] == 10
285
289
  assert row["reasoning_tokens"] == 5
286
- assert row["response_text"] == "gemini done"
290
+ assert captured_response(row)["content"] == "gemini done"
287
291
  assert row["request_id"] == "resp_g_1"
288
292
  assert row["sdk_version"] == metergraph.__version__
289
293
  _capture.set_runtime(None)
@@ -331,7 +335,7 @@ def test_wrap_google_stream_takes_usage_from_cumulative_last_chunk(tmp_path):
331
335
  assert row["input_tokens"] == 100
332
336
  assert row["output_tokens"] == 20
333
337
  assert row["cache_read_tokens"] == 10
334
- assert row["response_text"] == "partial"
338
+ assert captured_response(row)["content"] == "partial"
335
339
  _capture.set_runtime(None)
336
340
 
337
341
 
@@ -376,11 +380,11 @@ def test_wrap_google_patches_async_models(tmp_path):
376
380
  assert len(asyncio.run(run())) == 1
377
381
  assert rows.rows[0]["provider"] == "google"
378
382
  assert rows.rows[0]["endpoint"] == "models.generate_content"
379
- assert rows.rows[0]["response_text"] == "gemini async done"
383
+ assert captured_response(rows.rows[0])["content"] == "gemini async done"
380
384
  assert rows.rows[1]["endpoint"] == "models.generate_content.stream"
381
385
  assert rows.rows[1]["stream"] is True
382
386
  assert rows.rows[1]["input_tokens"] == 100
383
- assert rows.rows[1]["response_text"] == "gemini async stream"
387
+ assert captured_response(rows.rows[1])["content"] == "gemini async stream"
384
388
  _capture.set_runtime(None)
385
389
 
386
390
 
@@ -550,6 +554,55 @@ def test_anthropic_response_tool_use_is_requested_not_replayable(tmp_path):
550
554
  "idempotency": "non_idempotent",
551
555
  }
552
556
  ]
557
+ assert captured_response(rows.rows[0])["tool_calls"] == rows.rows[0]["tool_calls"]
558
+
559
+
560
+ def test_gemini_response_normalizes_function_calls(tmp_path):
561
+ rows = Rows()
562
+ runtime = Runtime(rows, Options(app_root=str(tmp_path)))
563
+ call = runtime.call_state(
564
+ "google",
565
+ "models.generate_content",
566
+ {
567
+ "model": "gemini-test",
568
+ "contents": [{"role": "user", "parts": [{"text": "Find order"}]}],
569
+ "tools": [{"name": "lookup_order"}],
570
+ },
571
+ )
572
+ result = SimpleNamespace(
573
+ candidates=[
574
+ SimpleNamespace(
575
+ content=SimpleNamespace(
576
+ parts=[
577
+ SimpleNamespace(
578
+ functionCall=SimpleNamespace(
579
+ id="call_g_1",
580
+ name="lookup_order",
581
+ args={"order_id": "ord_g_1"},
582
+ )
583
+ )
584
+ ]
585
+ )
586
+ )
587
+ ],
588
+ usage_metadata=SimpleNamespace(
589
+ prompt_token_count=10, candidates_token_count=2
590
+ ),
591
+ )
592
+
593
+ call.finish(result)
594
+
595
+ assert rows.rows[0]["tool_calls"] == [
596
+ {
597
+ "call_id": "call_g_1",
598
+ "name": "lookup_order",
599
+ "arguments": {"order_id": "ord_g_1"},
600
+ "result": None,
601
+ "status": "requested",
602
+ "idempotency": "non_idempotent",
603
+ }
604
+ ]
605
+ assert captured_response(rows.rows[0])["tool_calls"] == rows.rows[0]["tool_calls"]
553
606
 
554
607
 
555
608
  def test_openai_batch_output_file_captures_each_inference(tmp_path):
@@ -595,7 +648,7 @@ def test_openai_batch_output_file_captures_each_inference(tmp_path):
595
648
  assert row["batch_custom_id"] == "ticket-1"
596
649
  assert row["input_tokens"] == 11
597
650
  assert row["output_tokens"] == 3
598
- assert row["response_text"] == "batch answer"
651
+ assert captured_response(row)["content"] == "batch answer"
599
652
  assert row["request_id"] == "req_batch_1"
600
653
 
601
654
  # Re-reading the same output file in one process cannot double count it.
@@ -653,11 +706,11 @@ def test_anthropic_batch_results_capture_usage_without_changing_iteration(tmp_pa
653
706
  assert row["cache_write_5m_tokens"] == 2
654
707
  assert row["cache_write_1h_tokens"] == 5
655
708
  assert "cost_usd" not in row
656
- assert row["response_text"] == "anthropic batch answer"
709
+ assert captured_response(row)["content"] == "anthropic batch answer"
657
710
  _capture.set_runtime(None)
658
711
 
659
712
 
660
- def test_content_defaults_off_and_route_can_override_global_consent(tmp_path):
713
+ def test_content_defaults_on_and_route_can_override_capture(tmp_path):
661
714
  rows = Rows()
662
715
  runtime = Runtime(rows, Options(app_root=str(tmp_path)))
663
716
  call = runtime.call_state(
@@ -665,9 +718,9 @@ def test_content_defaults_off_and_route_can_override_global_consent(tmp_path):
665
718
  )
666
719
  call.finish(response("private output"))
667
720
 
668
- assert rows.rows[0]["content_opted_in"] is False
669
- assert rows.rows[0]["request_json"] is None
670
- assert rows.rows[0]["response_text"] is None
721
+ assert rows.rows[0]["content_opted_in"] is True
722
+ assert "private" in rows.rows[0]["request_json"]
723
+ assert captured_response(rows.rows[0])["content"] == "private output"
671
724
 
672
725
  _capture.set_runtime(runtime)
673
726
 
@@ -682,7 +735,7 @@ def test_content_defaults_off_and_route_can_override_global_consent(tmp_path):
682
735
 
683
736
  assert rows.rows[1]["content_opted_in"] is True
684
737
  assert "consented input" in rows.rows[1]["request_json"]
685
- assert rows.rows[1]["response_text"] == "consented output"
738
+ assert captured_response(rows.rows[1])["content"] == "consented output"
686
739
  _capture.set_runtime(None)
687
740
 
688
741
 
@@ -706,6 +759,99 @@ def test_route_opt_out_overrides_global_content_capture(tmp_path):
706
759
  _capture.set_runtime(None)
707
760
 
708
761
 
762
+ def test_trace_groups_spans_propagates_ids_and_supports_decorators(tmp_path):
763
+ rows = Rows()
764
+ runtime = Runtime(rows, Options(app_root=str(tmp_path)))
765
+ _capture.set_runtime(runtime)
766
+
767
+ class Responses:
768
+ def create(self, **kwargs):
769
+ return response(kwargs["input"])
770
+
771
+ client = SimpleNamespace(responses=Responses())
772
+ metergraph.wrap(client, provider="openai")
773
+ manual_trace_id = "a" * 32
774
+ parent_span_id = "b" * 16
775
+
776
+ with metergraph.trace(
777
+ "checkout", trace_id=manual_trace_id, parent_span_id=parent_span_id
778
+ ):
779
+ client.responses.create(model="test", input="first")
780
+ with metergraph.trace("nested-reuses-active"):
781
+ client.responses.create(model="test", input="second")
782
+ with metergraph.trace("explicit-fork", trace_id="c" * 32):
783
+ client.responses.create(model="test", input="forked")
784
+
785
+ @metergraph.trace("async-checkout")
786
+ async def traced_async():
787
+ await asyncio.sleep(0)
788
+ client.responses.create(model="test", input="third")
789
+ client.responses.create(model="test", input="fourth")
790
+
791
+ asyncio.run(traced_async())
792
+
793
+ assert {row["trace_id"] for row in rows.rows[:2]} == {manual_trace_id}
794
+ assert {row["trace_name"] for row in rows.rows[:2]} == {"checkout"}
795
+ assert {row["parent_span_id"] for row in rows.rows[:2]} == {parent_span_id}
796
+ assert len({row["span_id"] for row in rows.rows[:2]}) == 2
797
+ assert rows.rows[2]["trace_id"] == "c" * 32
798
+ assert rows.rows[2]["trace_name"] == "explicit-fork"
799
+ assert rows.rows[3]["trace_id"] == rows.rows[4]["trace_id"]
800
+ assert rows.rows[3]["trace_name"] == "async-checkout"
801
+ assert rows.rows[3]["trace_id"] != manual_trace_id
802
+ _capture.set_runtime(None)
803
+
804
+
805
+ def test_trace_capture_override_scrubbing_redaction_and_utf8_limits(tmp_path):
806
+ rows = Rows()
807
+
808
+ def redact(value, kind):
809
+ return value.replace("customer-secret", f"<redacted-{kind}>")
810
+
811
+ runtime = Runtime(
812
+ rows,
813
+ Options(
814
+ app_root=str(tmp_path),
815
+ capture_text=True,
816
+ redact=redact,
817
+ text_max_bytes=100 * 1024,
818
+ ),
819
+ )
820
+ hidden = runtime.call_state(
821
+ "openai", "responses", {"model": "test", "input": "hidden"}
822
+ )
823
+ with metergraph.trace("sensitive", capture_text=False):
824
+ # CaptureContext is read when the call starts.
825
+ hidden = runtime.call_state(
826
+ "openai", "responses", {"model": "test", "input": "hidden"}
827
+ )
828
+ hidden.finish(response("hidden output"))
829
+ assert rows.rows[0]["content_opted_in"] is False
830
+ assert rows.rows[0]["request_json"] is None
831
+ assert rows.rows[0]["response_text"] is None
832
+
833
+ call = runtime.call_state(
834
+ "openai",
835
+ "responses",
836
+ {
837
+ "model": "test",
838
+ "authorization": "Bearer provider-secret",
839
+ "headers": {"x-api-key": "provider-secret"},
840
+ "input": "customer-secret" + ("ü" * 80_000),
841
+ },
842
+ )
843
+ call.finish(response("customer-secret" + ("é" * 80_000)))
844
+ row = rows.rows[1]
845
+ assert "provider-secret" not in row["request_json"]
846
+ assert "customer-secret" not in row["request_json"]
847
+ assert "customer-secret" not in row["response_text"]
848
+ assert len(row["request_json"].encode()) <= 100 * 1024
849
+ assert len(row["response_text"].encode()) <= 100 * 1024
850
+ assert row["text_truncated"] is True
851
+ assert row["request_json"].endswith("<metergraph:truncated>")
852
+ assert row["response_text"].endswith("<metergraph:truncated>")
853
+
854
+
709
855
  def test_wrap_async_errors_are_recorded_and_original_error_is_raised(tmp_path):
710
856
  rows = Rows()
711
857
  _capture.set_runtime(
@@ -730,6 +876,10 @@ def test_wrap_async_errors_are_recorded_and_original_error_is_raised(tmp_path):
730
876
  asyncio.run(run())
731
877
  assert rows.rows[0]["error"] is True
732
878
  assert rows.rows[0]["error_type"] == "ValueError"
879
+ assert captured_response(rows.rows[0])["error"] == {
880
+ "type": "ValueError",
881
+ "message": "provider down",
882
+ }
733
883
  _capture.set_runtime(None)
734
884
 
735
885
 
@@ -747,6 +897,44 @@ def test_stream_records_ttft_and_final_usage(tmp_path):
747
897
  choices=[SimpleNamespace(delta=SimpleNamespace(content="hi"))],
748
898
  usage=None,
749
899
  ),
900
+ SimpleNamespace(
901
+ choices=[
902
+ SimpleNamespace(
903
+ delta=SimpleNamespace(
904
+ content=None,
905
+ tool_calls=[
906
+ SimpleNamespace(
907
+ index=0,
908
+ id="call_1",
909
+ function=SimpleNamespace(
910
+ name="lookup", arguments='{"id":'
911
+ ),
912
+ )
913
+ ],
914
+ )
915
+ )
916
+ ],
917
+ usage=None,
918
+ ),
919
+ SimpleNamespace(
920
+ choices=[
921
+ SimpleNamespace(
922
+ delta=SimpleNamespace(
923
+ content=None,
924
+ tool_calls=[
925
+ SimpleNamespace(
926
+ index=0,
927
+ id=None,
928
+ function=SimpleNamespace(
929
+ name=None, arguments='"ord_1"}'
930
+ ),
931
+ )
932
+ ],
933
+ )
934
+ )
935
+ ],
936
+ usage=None,
937
+ ),
750
938
  SimpleNamespace(
751
939
  choices=[],
752
940
  usage=SimpleNamespace(
@@ -767,14 +955,22 @@ def test_stream_records_ttft_and_final_usage(tmp_path):
767
955
  )
768
956
  # The SDK-added OpenAI usage-only chunk is consumed for metering but is
769
957
  # not exposed to an application that did not ask for it.
770
- assert len(chunks) == 1
958
+ assert len(chunks) == 3
771
959
  assert rows.rows[0]["stream"] is True
772
960
  assert rows.rows[0]["ttft_ms"] is not None
773
961
  assert rows.rows[0]["input_tokens"] == 2
774
962
  assert rows.rows[0]["cache_read_tokens"] == 1
775
963
  assert rows.rows[0]["cache_write_tokens"] == 2
776
964
  assert "cost_usd" not in rows.rows[0]
777
- assert rows.rows[0]["response_text"] == "hi"
965
+ assert captured_response(rows.rows[0])["content"] == "hi"
966
+ assert captured_response(rows.rows[0])["tool_calls"][0] == {
967
+ "call_id": "call_1",
968
+ "name": "lookup",
969
+ "arguments": {"id": "ord_1"},
970
+ "result": None,
971
+ "status": "requested",
972
+ "idempotency": "non_idempotent",
973
+ }
778
974
  assert rows.rows[0]["request_json"].find("include_usage") >= 0
779
975
  _capture.set_runtime(None)
780
976
 
@@ -829,7 +1025,7 @@ def test_async_stream_awaits_anthropic_final_message(tmp_path):
829
1025
  assert rows.rows[0]["cache_write_5m_tokens"] == 2
830
1026
  assert rows.rows[0]["cache_write_1h_tokens"] == 3
831
1027
  assert "cost_usd" not in rows.rows[0]
832
- assert rows.rows[0]["response_text"] == "ok"
1028
+ assert captured_response(rows.rows[0])["content"] == "ok"
833
1029
  _capture.set_runtime(None)
834
1030
 
835
1031
 
File without changes