agentx-python 0.8.26__py3-none-any.whl → 0.8.28__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -14,8 +14,16 @@ from __future__ import annotations
14
14
  import asyncio
15
15
  import inspect
16
16
  import json
17
+ import logging
18
+ import threading
19
+ import time
17
20
  from typing import Any, Callable, Dict, Optional
18
21
 
22
+ logger = logging.getLogger(__name__)
23
+
24
+ # Sentinel for finish_llm_call's `active_span`: "not passed" is distinct from "passed None".
25
+ _UNSET: Any = object()
26
+
19
27
  from agentx.tracing.tracer import Tracer, _safe_serialize
20
28
 
21
29
 
@@ -107,6 +115,8 @@ def finish_llm_call(
107
115
  cache_read_tokens: Optional[int] = None,
108
116
  cache_write_tokens: Optional[int] = None,
109
117
  tool_definitions: Optional[list] = None,
118
+ call_metadata: Optional[Dict[str, Any]] = None,
119
+ active_span: Any = _UNSET,
110
120
  ) -> None:
111
121
  """
112
122
  Close out one raw-client LLM call - shared by the ``on_finish``/exit
@@ -127,7 +137,11 @@ def finish_llm_call(
127
137
  if tool_definitions:
128
138
  metadata = {**(metadata or {}), "tools": tool_definitions}
129
139
 
130
- active_span = tracer.current_span
140
+ # The parent is the span that was active when the CALL was made. Streaming patches pass it
141
+ # explicitly: a stream finalizes later (exhaustion, close, or garbage collection), by which
142
+ # time a different span may be active, and the call must not be grafted onto it.
143
+ if active_span is _UNSET:
144
+ active_span = tracer.current_span
131
145
  if active_span is not None:
132
146
  # The definitions describe the whole call's toolbox - attach them to the enclosing
133
147
  # span's metadata (first capture wins) so the ROOT trace carries them for the
@@ -150,16 +164,25 @@ def finish_llm_call(
150
164
  output_tokens=output_tokens,
151
165
  cache_read_tokens=cache_read_tokens,
152
166
  cache_write_tokens=cache_write_tokens,
167
+ metadata=call_metadata,
153
168
  )
154
169
  return
155
170
 
156
171
  # A patched provider call outside any active span becomes its own root trace - it is a bare
157
172
  # model call, so stamp it "llm" rather than leaving the kind unset.
158
173
  span = tracer.trace(
159
- name, metadata=metadata, framework=framework, model=model, session_id=session_id, span_kind="llm"
174
+ name,
175
+ metadata={**(metadata or {}), **(call_metadata or {})} if (metadata or call_metadata) else None,
176
+ framework=framework,
177
+ model=model,
178
+ session_id=session_id,
179
+ span_kind="llm",
160
180
  )
161
181
  span.__enter__()
162
182
  span._start = start_t
183
+ # The call ended at end_t (a stream's last chunk), not at whatever later moment this
184
+ # runs - __exit__ honors the override instead of measuring to time.time().
185
+ span._end_override = end_t
163
186
  span.input = input_repr
164
187
  span.output = output
165
188
  if error:
@@ -173,3 +196,251 @@ def finish_llm_call(
173
196
  if cache_write_tokens:
174
197
  span._cache_write_tokens = cache_write_tokens
175
198
  span.__exit__(None, None, None)
199
+
200
+
201
+ # ---------------------------------------------------------------------------
202
+ # Streaming: wrap a provider's chunk stream so the trace is built from what
203
+ # was actually streamed, without touching the caller's consumption of it.
204
+ # ---------------------------------------------------------------------------
205
+
206
+ class StreamAccumulator:
207
+ """
208
+ What a streaming patch feeds each chunk into. Subclasses collect the
209
+ provider-specific pieces (text deltas, tool-call deltas, the usage block
210
+ that only arrives on the final chunk) and hand back the finished picture
211
+ in ``result()``.
212
+ """
213
+
214
+ def feed(self, chunk: Any) -> None: # pragma: no cover - interface
215
+ raise NotImplementedError
216
+
217
+ def result(self) -> Dict[str, Any]: # pragma: no cover - interface
218
+ raise NotImplementedError
219
+
220
+
221
+ class TracedStream:
222
+ """
223
+ Transparent proxy over a provider ``Stream``/``AsyncStream``: iterates the
224
+ real object, feeds every chunk to the accumulator, and calls ``on_finish``
225
+ exactly once when the stream is exhausted, raises, is closed (``close()``,
226
+ ``with``/``async with`` exit), or is dropped part-way and garbage
227
+ collected - so an abandoned stream still records what it streamed.
228
+
229
+ Latency is measured to the LAST chunk (the response as the caller saw it),
230
+ and the time to the FIRST chunk is reported separately as
231
+ ``time_to_first_token_ms`` - the two numbers a streaming call is judged by.
232
+
233
+ Attribute access falls through to the wrapped stream (``.response``,
234
+ provider helpers), and ``__iter__``/``__aiter__`` return ``self`` so early
235
+ ``break`` leaves no half-driven generator behind. It is a proxy, not a
236
+ subclass: ``isinstance(stream, openai.Stream)`` is False and ``repr()``
237
+ shows the proxy - branch on ``stream=True`` in your own code, not on type.
238
+
239
+ Tracing never breaks the caller: a failure while building or sending the
240
+ trace is logged and swallowed, and the stream's own iteration/close
241
+ semantics are untouched.
242
+ """
243
+
244
+ def __init__(
245
+ self,
246
+ stream: Any,
247
+ accumulator: StreamAccumulator,
248
+ on_finish: Callable[[Dict[str, Any], Optional[str]], None],
249
+ ) -> None:
250
+ self._stream = stream
251
+ self._accumulator = accumulator
252
+ self._on_finish = on_finish
253
+ self._done = False
254
+ # A watchdog close() racing the reader's StopIteration must not finalize twice.
255
+ self._done_lock = threading.Lock()
256
+ self._first_chunk_t: Optional[float] = None
257
+ self._last_chunk_t: Optional[float] = None
258
+ self._sync_iter: Any = None
259
+ self._async_iter: Any = None
260
+
261
+ # -- bookkeeping ---------------------------------------------------------
262
+
263
+ def _observe(self, chunk: Any) -> None:
264
+ now = time.time()
265
+ if self._first_chunk_t is None:
266
+ self._first_chunk_t = now
267
+ self._last_chunk_t = now
268
+ try:
269
+ self._accumulator.feed(chunk)
270
+ except Exception:
271
+ # A malformed chunk must never break the caller's stream; it just
272
+ # goes uncounted in the trace.
273
+ pass
274
+
275
+ def _finish(self, error: Optional[str]) -> None:
276
+ with self._done_lock:
277
+ if self._done:
278
+ return
279
+ self._done = True
280
+ try:
281
+ result = self._accumulator.result()
282
+ except Exception:
283
+ result = {}
284
+ start_t = result.pop("_start_t", None)
285
+ result["time_to_first_token_ms"] = (
286
+ int((self._first_chunk_t - start_t) * 1000) if self._first_chunk_t is not None and start_t is not None else None
287
+ )
288
+ # The response "ended" at its last chunk, not at whatever later moment the caller closed
289
+ # or dropped the stream - that is the latency the user experienced.
290
+ result["end_t"] = self._last_chunk_t if self._last_chunk_t is not None else time.time()
291
+ try:
292
+ self._on_finish(result, error)
293
+ except Exception:
294
+ # Building or sending the trace failed. The caller's stream ended normally and must
295
+ # see it end normally - tracing is never allowed to raise into inference code.
296
+ logger.debug("Streamed call could not be traced", exc_info=True)
297
+
298
+ @property
299
+ def first_chunk_at(self) -> Optional[float]:
300
+ return self._first_chunk_t
301
+
302
+ # -- sync iteration ------------------------------------------------------
303
+
304
+ def __iter__(self) -> "TracedStream":
305
+ return self
306
+
307
+ def __next__(self) -> Any:
308
+ if self._sync_iter is None:
309
+ self._sync_iter = iter(self._stream)
310
+ try:
311
+ chunk = next(self._sync_iter)
312
+ except StopIteration:
313
+ self._finish(None)
314
+ raise
315
+ except Exception as exc:
316
+ self._finish(str(exc))
317
+ raise
318
+ except BaseException:
319
+ # KeyboardInterrupt / GeneratorExit: a cancellation, not the provider failing -
320
+ # record what streamed so far without inventing an error message.
321
+ self._finish(None)
322
+ raise
323
+ self._observe(chunk)
324
+ return chunk
325
+
326
+ # -- async iteration -----------------------------------------------------
327
+
328
+ def __aiter__(self) -> "TracedStream":
329
+ return self
330
+
331
+ async def __anext__(self) -> Any:
332
+ if self._async_iter is None:
333
+ self._async_iter = self._stream.__aiter__()
334
+ try:
335
+ chunk = await self._async_iter.__anext__()
336
+ except StopAsyncIteration:
337
+ self._finish(None)
338
+ raise
339
+ except Exception as exc:
340
+ self._finish(str(exc))
341
+ raise
342
+ except BaseException:
343
+ self._finish(None)
344
+ raise
345
+ self._observe(chunk)
346
+ return chunk
347
+
348
+ # -- context managers / close --------------------------------------------
349
+
350
+ def __enter__(self) -> "TracedStream":
351
+ enter = getattr(self._stream, "__enter__", None)
352
+ if enter is not None:
353
+ enter()
354
+ return self
355
+
356
+ def __exit__(self, exc_type, exc_val, tb) -> Any:
357
+ exit_ = getattr(self._stream, "__exit__", None)
358
+ result = exit_(exc_type, exc_val, tb) if exit_ is not None else None
359
+ self._finish(str(exc_val) if exc_val else None)
360
+ return result
361
+
362
+ async def __aenter__(self) -> "TracedStream":
363
+ enter = getattr(self._stream, "__aenter__", None)
364
+ if enter is not None:
365
+ await enter()
366
+ return self
367
+
368
+ async def __aexit__(self, exc_type, exc_val, tb) -> Any:
369
+ exit_ = getattr(self._stream, "__aexit__", None)
370
+ result = await exit_(exc_type, exc_val, tb) if exit_ is not None else None
371
+ self._finish(str(exc_val) if exc_val else None)
372
+ return result
373
+
374
+ def close(self) -> None:
375
+ close = getattr(self._stream, "close", None)
376
+ try:
377
+ if close is not None:
378
+ result = close()
379
+ if inspect.isawaitable(result):
380
+ # openai's AsyncStream spells its close `async def close()`. A sync close()
381
+ # on it (an easy slip inside async code) would drop the coroutine and leak
382
+ # the connection; run it on the loop when there is one, else at least don't
383
+ # leave an un-awaited coroutine behind.
384
+ try:
385
+ asyncio.get_running_loop().create_task(result)
386
+ except RuntimeError:
387
+ result.close() # type: ignore[union-attr]
388
+ logger.warning("close() called on an async stream outside an event loop - use aclose()")
389
+ finally:
390
+ self._finish(None)
391
+
392
+ async def aclose(self) -> None:
393
+ # openai's AsyncStream spells its close as `async def close()`; httpx-style streams
394
+ # spell it `aclose()`. Await whichever one answers with an awaitable.
395
+ close = getattr(self._stream, "aclose", None) or getattr(self._stream, "close", None)
396
+ try:
397
+ if close is not None:
398
+ result = close()
399
+ if inspect.isawaitable(result):
400
+ await result
401
+ finally:
402
+ self._finish(None)
403
+
404
+ def __getattr__(self, item: str) -> Any:
405
+ # Only public attributes delegate. Private names must resolve on the proxy itself, or a
406
+ # half-constructed instance (no _stream yet) would recurse forever looking for it.
407
+ if item.startswith("_"):
408
+ raise AttributeError(item)
409
+ return getattr(self._stream, item)
410
+
411
+ def __del__(self) -> None:
412
+ # Best effort only: a stream the caller stopped reading and dropped still records the
413
+ # chunks it did see. Never raises - a destructor exception is unactionable noise.
414
+ try:
415
+ self._finish(None)
416
+ except Exception:
417
+ pass
418
+
419
+
420
+ def trace_stream(
421
+ result: Any,
422
+ accumulator: StreamAccumulator,
423
+ on_finish: Callable[[Dict[str, Any], Optional[str]], None],
424
+ ) -> Any:
425
+ """
426
+ Wrap the value a patched ``create(..., stream=True)`` returned. A sync
427
+ client hands back the stream object directly; an async client hands back
428
+ a coroutine that resolves to it, so the wrapping is deferred until the
429
+ real stream exists - the caller's ``await`` is unchanged either way.
430
+ """
431
+ if asyncio.iscoroutine(result) or inspect.isawaitable(result):
432
+ return _await_and_wrap(result, accumulator, on_finish)
433
+ return TracedStream(result, accumulator, on_finish)
434
+
435
+
436
+ async def _await_and_wrap(
437
+ awaitable: Any,
438
+ accumulator: StreamAccumulator,
439
+ on_finish: Callable[[Dict[str, Any], Optional[str]], None],
440
+ ) -> Any:
441
+ try:
442
+ stream = await awaitable
443
+ except Exception as exc:
444
+ on_finish({}, str(exc))
445
+ raise
446
+ return TracedStream(stream, accumulator, on_finish)
@@ -13,6 +13,12 @@ Usage::
13
13
 
14
14
  Works with both ``anthropic.Anthropic`` and ``anthropic.AsyncAnthropic`` clients.
15
15
 
16
+ Both streaming shapes are traced: the ``client.messages.stream(...)`` helper
17
+ (a context manager with ``get_final_message()``) and the raw
18
+ ``messages.create(..., stream=True)`` event stream, which is wrapped in a
19
+ transparent proxy that assembles the reply, tool-use blocks, and token usage
20
+ from the events as the caller consumes them.
21
+
16
22
  Requires: ``pip install "agentx-python[anthropic]"``
17
23
  """
18
24
  from __future__ import annotations
@@ -22,7 +28,13 @@ import time
22
28
  from typing import Any, Dict, Optional, Tuple
23
29
 
24
30
  from agentx.tracing.tracer import Tracer, _safe_serialize
25
- from agentx.integrations._traced_call import capture_tool_definitions, call_and_trace, finish_llm_call
31
+ from agentx.integrations._traced_call import (
32
+ StreamAccumulator,
33
+ capture_tool_definitions,
34
+ call_and_trace,
35
+ finish_llm_call,
36
+ trace_stream,
37
+ )
26
38
 
27
39
 
28
40
  def _extract_output_text(response: Any) -> Optional[str]:
@@ -92,6 +104,85 @@ def _extract_usage_tokens(
92
104
  return input_tokens, output_tokens, cache_read, cache_creation
93
105
 
94
106
 
107
+ class _MessageEventStreamAccumulator(StreamAccumulator):
108
+ """
109
+ Rebuild a ``Message`` from the raw ``create(stream=True)`` event sequence:
110
+ ``message_start`` carries the input-side usage, ``content_block_start`` opens
111
+ a text or tool_use block, ``content_block_delta`` appends ``text_delta`` /
112
+ ``input_json_delta`` fragments to it, ``message_delta`` carries the
113
+ output-token count. Token accounting mirrors ``_extract_usage_tokens``.
114
+ """
115
+
116
+ def __init__(self, start_t: float) -> None:
117
+ self._start_t = start_t
118
+ self._blocks: Dict[int, Dict[str, Any]] = {}
119
+ self._input_tokens: Optional[int] = None
120
+ self._output_tokens: Optional[int] = None
121
+ self._cache_read: Optional[int] = None
122
+ self._cache_write: Optional[int] = None
123
+ self._model: Optional[str] = None
124
+
125
+ def feed(self, event: Any) -> None:
126
+ event_type = getattr(event, "type", None)
127
+ if event_type == "message_start":
128
+ message = getattr(event, "message", None)
129
+ self._model = getattr(message, "model", None) or self._model
130
+ input_tokens, output_tokens, cache_read, cache_write = _extract_usage_tokens(getattr(message, "usage", None))
131
+ self._input_tokens = input_tokens
132
+ self._cache_read = cache_read
133
+ self._cache_write = cache_write
134
+ if output_tokens:
135
+ self._output_tokens = output_tokens
136
+ elif event_type == "content_block_start":
137
+ index = getattr(event, "index", 0) or 0
138
+ block = getattr(event, "content_block", None)
139
+ self._blocks[index] = {
140
+ "type": getattr(block, "type", None),
141
+ "name": getattr(block, "name", None),
142
+ "text": [getattr(block, "text", None) or ""] if getattr(block, "type", None) == "text" else [],
143
+ "json": [],
144
+ }
145
+ elif event_type == "content_block_delta":
146
+ index = getattr(event, "index", 0) or 0
147
+ delta = getattr(event, "delta", None)
148
+ entry = self._blocks.setdefault(index, {"type": None, "name": None, "text": [], "json": []})
149
+ delta_type = getattr(delta, "type", None)
150
+ if delta_type == "text_delta":
151
+ entry["type"] = entry["type"] or "text"
152
+ entry["text"].append(getattr(delta, "text", None) or "")
153
+ elif delta_type == "input_json_delta":
154
+ entry["type"] = entry["type"] or "tool_use"
155
+ entry["json"].append(getattr(delta, "partial_json", None) or "")
156
+ elif event_type == "message_delta":
157
+ usage = getattr(event, "usage", None)
158
+ output_tokens = getattr(usage, "output_tokens", None) if usage is not None else None
159
+ if output_tokens is not None:
160
+ self._output_tokens = output_tokens
161
+
162
+ def result(self) -> Dict[str, Any]:
163
+ texts = []
164
+ tool_calls = []
165
+ for _, block in sorted(self._blocks.items()):
166
+ if block["type"] == "text":
167
+ text = "".join(block["text"])
168
+ if text:
169
+ texts.append(text)
170
+ elif block["type"] == "tool_use":
171
+ tool_calls.append(f"{block['name'] or 'unknown'}({''.join(block['json'])})")
172
+ output: Optional[str] = "\n".join(texts) if texts else None
173
+ if output is None and tool_calls:
174
+ output = "[tool call] " + ", ".join(tool_calls)
175
+ return {
176
+ "_start_t": self._start_t,
177
+ "output": output,
178
+ "model": self._model,
179
+ "input_tokens": self._input_tokens,
180
+ "output_tokens": self._output_tokens,
181
+ "cache_read_tokens": self._cache_read,
182
+ "cache_write_tokens": self._cache_write,
183
+ }
184
+
185
+
95
186
  def patch_anthropic_client(
96
187
  client: Any,
97
188
  tracer: Tracer,
@@ -140,6 +231,43 @@ def _patch_create(
140
231
 
141
232
  input_repr = _safe_serialize(input_messages)
142
233
 
234
+ if kwargs.get("stream"):
235
+ # Parent fixed at call time - see openai.py's patched_create for why.
236
+ parent = tracer.current_span
237
+
238
+ def on_stream_finish(collected: Dict[str, Any], error: Optional[str]) -> None:
239
+ ttft = collected.get("time_to_first_token_ms")
240
+ call_metadata: Dict[str, Any] = {"streaming": True}
241
+ if ttft is not None:
242
+ call_metadata["timeToFirstTokenMs"] = ttft
243
+ finish_llm_call(
244
+ tracer,
245
+ name=name,
246
+ framework="anthropic",
247
+ metadata=metadata,
248
+ call_metadata=call_metadata,
249
+ active_span=parent,
250
+ session_id=session_id,
251
+ start_t=start_t,
252
+ end_t=collected.get("end_t") or time.time(),
253
+ input_repr=input_repr,
254
+ output=collected.get("output"),
255
+ model=collected.get("model") or model,
256
+ input_tokens=collected.get("input_tokens"),
257
+ output_tokens=collected.get("output_tokens"),
258
+ cache_read_tokens=collected.get("cache_read_tokens"),
259
+ cache_write_tokens=collected.get("cache_write_tokens"),
260
+ error=error,
261
+ tool_definitions=tool_definitions,
262
+ )
263
+
264
+ try:
265
+ result = original(*args, **kwargs)
266
+ except Exception as exc:
267
+ on_stream_finish({}, str(exc))
268
+ raise
269
+ return trace_stream(result, _MessageEventStreamAccumulator(start_t), on_stream_finish)
270
+
143
271
  def on_finish(response: Optional[Any], error: Optional[str]) -> None:
144
272
  end_t = time.time()
145
273
  output = None
@@ -198,8 +326,9 @@ def _patch_stream(
198
326
  # only shows up in whether `with`/`async with` and
199
327
  # `get_final_message()` are used, handled inside `_TracedStream`.
200
328
  start_t = time.time()
329
+ parent = tracer.current_span
201
330
  ctx = original_stream(*args, **kwargs)
202
- input_repr = _safe_serialize(_prepend_system(kwargs.get("messages"), kwargs.get("system")))
331
+ input_repr = _safe_serialize(_prepend_system(kwargs.get("messages") or (args[0] if args else None), kwargs.get("system")))
203
332
  model = kwargs.get("model")
204
333
  tool_definitions = capture_tool_definitions(kwargs.get("tools"))
205
334
 
@@ -237,38 +366,72 @@ def _patch_stream(
237
366
  )
238
367
 
239
368
  class _TracedStream:
240
- """Thin wrapper that records timing when the stream context exits."""
369
+ """
370
+ Thin wrapper that records the final message when the stream context exits.
371
+ ``ctx`` is the SDK's stream *manager*; the ``MessageStream`` it yields on enter is
372
+ what carries ``get_final_message()``, and it must be read BEFORE the manager's exit
373
+ closes it - reading it off the manager after close silently yielded no output.
374
+ """
375
+
376
+ _inner: Any = None
377
+ _sent: bool = False
378
+
379
+ # What streamed so far, WITHOUT draining the rest of the response: the SDK's
380
+ # get_final_message() calls until_done(), which would turn an early `break` into a
381
+ # blocking read of every remaining token. The snapshot is the final message once the
382
+ # stream was consumed, and honestly partial when the caller stopped early.
383
+ def _snapshot(self_inner):
384
+ inner = self_inner._inner
385
+ if inner is None:
386
+ return None
387
+ try:
388
+ return getattr(inner, "current_message_snapshot", None)
389
+ except Exception:
390
+ return None
391
+
392
+ def _send_once(self_inner, end_t: float, error: Optional[str], snapshot: Any) -> None:
393
+ if self_inner._sent:
394
+ return
395
+ self_inner._sent = True
396
+ try:
397
+ build_and_send(end_t, error, snapshot)
398
+ except Exception:
399
+ pass # tracing never raises into the caller
241
400
 
242
401
  def __enter__(self_inner):
243
- return ctx.__enter__()
402
+ self_inner._inner = ctx.__enter__()
403
+ return self_inner._inner
244
404
 
245
405
  def __exit__(self_inner, exc_type, exc_val, tb):
246
- result = ctx.__exit__(exc_type, exc_val, tb)
247
406
  end_t = time.time()
248
407
  error = str(exc_val) if exc_val else None
249
- final_message = None
408
+ # Snapshot BEFORE the manager closes the stream (the earlier bug read it after).
409
+ snapshot = self_inner._snapshot()
250
410
  try:
251
- final_message = ctx.get_final_message()
252
- except Exception:
253
- pass
254
- build_and_send(end_t, error, final_message)
255
- return result
411
+ return ctx.__exit__(exc_type, exc_val, tb)
412
+ finally:
413
+ self_inner._send_once(end_t, error, snapshot)
256
414
 
257
415
  async def __aenter__(self_inner):
258
- return await ctx.__aenter__()
416
+ self_inner._inner = await ctx.__aenter__()
417
+ return self_inner._inner
259
418
 
260
419
  async def __aexit__(self_inner, exc_type, exc_val, tb):
261
- result = await ctx.__aexit__(exc_type, exc_val, tb)
262
420
  end_t = time.time()
263
421
  error = str(exc_val) if exc_val else None
264
- final_message = None
422
+ snapshot = self_inner._snapshot()
423
+ try:
424
+ return await ctx.__aexit__(exc_type, exc_val, tb)
425
+ finally:
426
+ self_inner._send_once(end_t, error, snapshot)
427
+
428
+ def __del__(self_inner):
429
+ # A helper stream that was entered but never exited still records what it saw.
265
430
  try:
266
- raw = ctx.get_final_message()
267
- final_message = await raw if inspect.isawaitable(raw) else raw
431
+ if self_inner._inner is not None:
432
+ self_inner._send_once(time.time(), None, self_inner._snapshot())
268
433
  except Exception:
269
434
  pass
270
- build_and_send(end_t, error, final_message)
271
- return result
272
435
 
273
436
  def __iter__(self_inner):
274
437
  return iter(ctx)
@@ -27,8 +27,11 @@ Works with both ``openai.OpenAI`` and ``openai.AsyncOpenAI`` clients. Token
27
27
  usage comes straight off the response's OpenAI-shaped ``usage`` block; NIM
28
28
  reports no prompt-cache fields, so cache token counts stay unset.
29
29
 
30
- Streaming calls (``stream=True``) are passed through untouched and are not
31
- currently traced - same posture as ``patch_openai_client``, see its docstring.
30
+ Streaming calls (``stream=True``) are traced too, exactly as
31
+ ``patch_openai_client`` traces them: the stream is wrapped in a transparent
32
+ proxy that assembles the reply from the consumed chunks (token usage when the
33
+ endpoint sends it on the final chunk, e.g. with
34
+ ``stream_options={"include_usage": True}``).
32
35
 
33
36
  Requires: ``pip install "agentx-python[nvidia-nim]"`` (installs the ``openai``
34
37
  client package; there is no separate NIM SDK dependency).
@@ -56,8 +59,9 @@ def patch_nim_client(
56
59
  call with ``framework="nvidia-nim"``.
57
60
 
58
61
  The original method is still called and its return value passed through
59
- unchanged. Sync and async clients both work; ``stream=True`` calls pass
60
- through untraced. Patching is idempotent - and because it shares the guard
62
+ unchanged. Sync and async clients both work; ``stream=True`` calls are
63
+ traced through the same stream proxy as ``patch_openai_client``. Patching
64
+ is idempotent - and because it shares the guard
61
65
  with ``patch_openai_client``, whichever of the two patched a given client
62
66
  first wins (patch each client with the integration that matches where its
63
67
  ``base_url`` actually points).
@@ -17,8 +17,11 @@ Usage::
17
17
 
18
18
  Works with both ``openai.OpenAI`` and ``openai.AsyncOpenAI`` clients.
19
19
 
20
- Streaming calls (``stream=True``) are passed through untouched and are not
21
- currently traced - see ``patch_openai_client``'s docstring.
20
+ Streaming calls (``stream=True``) are traced too: the returned stream is
21
+ wrapped in a transparent proxy that assembles the reply from the chunks as the
22
+ caller consumes them, so the trace carries the full text, tool calls, and
23
+ (with ``stream_options={"include_usage": True}``) token usage, plus the time
24
+ to first token.
22
25
 
23
26
  Requires: ``pip install "agentx-python[openai]"``
24
27
  """
@@ -27,8 +30,21 @@ from __future__ import annotations
27
30
  import time
28
31
  from typing import Any, Dict, Optional, Tuple
29
32
 
33
+ import logging
34
+
30
35
  from agentx.tracing.tracer import Tracer, _safe_serialize
31
- from agentx.integrations._traced_call import capture_tool_definitions, call_and_trace, finish_llm_call
36
+
37
+ logger = logging.getLogger(__name__)
38
+ # Warn once per process, not per call: a streamed OpenAI call carries no usage unless the caller
39
+ # asked for it, and a silent zero would under-report every streaming app's spend.
40
+ _warned_stream_usage = False
41
+ from agentx.integrations._traced_call import (
42
+ StreamAccumulator,
43
+ capture_tool_definitions,
44
+ call_and_trace,
45
+ finish_llm_call,
46
+ trace_stream,
47
+ )
32
48
 
33
49
 
34
50
  def _extract_output_text(response: Any) -> Optional[str]:
@@ -75,6 +91,68 @@ def _extract_usage_tokens(usage: Any) -> Tuple[Optional[int], Optional[int], Opt
75
91
  return getattr(usage, "prompt_tokens", None), getattr(usage, "completion_tokens", None), cached_tokens
76
92
 
77
93
 
94
+ class _ChatCompletionStreamAccumulator(StreamAccumulator):
95
+ """
96
+ Rebuild a ``ChatCompletion``-shaped result from ``ChatCompletionChunk``s:
97
+ text deltas concatenate per choice, tool-call deltas merge by index (name
98
+ arrives once, arguments arrive as fragments), and the ``usage`` block -
99
+ present only on the final chunk, and only when the caller asked for it
100
+ with ``stream_options={"include_usage": True}`` - is kept when it appears.
101
+ """
102
+
103
+ def __init__(self, start_t: float) -> None:
104
+ self._start_t = start_t
105
+ self._texts: Dict[int, list] = {}
106
+ self._tool_calls: Dict[int, Dict[str, Any]] = {}
107
+ self._usage: Any = None
108
+ self._model: Optional[str] = None
109
+
110
+ def feed(self, chunk: Any) -> None:
111
+ usage = getattr(chunk, "usage", None)
112
+ if usage is not None:
113
+ self._usage = usage
114
+ model = getattr(chunk, "model", None)
115
+ if model and not self._model:
116
+ self._model = model
117
+ for choice in getattr(chunk, "choices", None) or []:
118
+ index = getattr(choice, "index", 0) or 0
119
+ delta = getattr(choice, "delta", None)
120
+ if delta is None:
121
+ continue
122
+ content = getattr(delta, "content", None)
123
+ if content:
124
+ self._texts.setdefault(index, []).append(content)
125
+ for tc in getattr(delta, "tool_calls", None) or []:
126
+ key = getattr(tc, "index", 0) or 0
127
+ entry = self._tool_calls.setdefault(key, {"name": None, "arguments": []})
128
+ fn = getattr(tc, "function", None)
129
+ fn_name = getattr(fn, "name", None) if fn is not None else None
130
+ fn_args = getattr(fn, "arguments", None) if fn is not None else None
131
+ if fn_name:
132
+ entry["name"] = fn_name
133
+ if fn_args:
134
+ entry["arguments"].append(fn_args)
135
+
136
+ def result(self) -> Dict[str, Any]:
137
+ texts = ["".join(parts) for _, parts in sorted(self._texts.items())]
138
+ output: Optional[str] = "\n".join(t for t in texts if t) or None
139
+ if output is None and self._tool_calls:
140
+ described = [
141
+ f"{entry['name'] or 'unknown'}({''.join(entry['arguments'])})"
142
+ for _, entry in sorted(self._tool_calls.items())
143
+ ]
144
+ output = "[tool call] " + ", ".join(described)
145
+ input_tokens, output_tokens, cache_read_tokens = _extract_usage_tokens(self._usage)
146
+ return {
147
+ "_start_t": self._start_t,
148
+ "output": output,
149
+ "model": self._model,
150
+ "input_tokens": input_tokens,
151
+ "output_tokens": output_tokens,
152
+ "cache_read_tokens": cache_read_tokens,
153
+ }
154
+
155
+
78
156
  def patch_openai_client(
79
157
  client: Any,
80
158
  tracer: Tracer,
@@ -92,12 +170,14 @@ def patch_openai_client(
92
170
  client's ``create()`` returns a coroutine, which is detected and awaited
93
171
  before the trace is built.
94
172
 
95
- Calls made with ``stream=True`` are passed through untouched and are not
96
- traced by this function: safely wrapping a (sync or async) chunk
97
- iterator without disrupting the caller's own consumption of it needs
98
- different handling than a single request/response call, so it's left
99
- unpatched rather than risking a partially-consumed or double-consumed
100
- stream for the caller.
173
+ Calls made with ``stream=True`` return a transparent proxy over the
174
+ provider's stream (see ``_traced_call.TracedStream``): iteration, ``with``,
175
+ ``close()`` and attribute access all pass through to the real stream, and
176
+ the trace is built from the chunks the caller actually consumed - text and
177
+ tool calls assembled from the deltas, token usage from the final chunk
178
+ when ``stream_options={"include_usage": True}`` was requested (OpenAI omits
179
+ usage from streams otherwise), latency to the last chunk, and the time to
180
+ first token in the trace metadata.
101
181
  """
102
182
  chat = getattr(client, "chat", None)
103
183
  completions = getattr(chat, "completions", None) if chat is not None else None
@@ -123,17 +203,57 @@ def _patch_chat_completions_create(
123
203
  return # already patched
124
204
 
125
205
  def patched_create(*args, **kwargs):
126
- if kwargs.get("stream"):
127
- # Not traced - see patch_openai_client's docstring. Passed
128
- # through completely untouched, sync or async.
129
- return original(*args, **kwargs)
130
-
131
206
  start_t = time.time()
132
207
  input_messages = kwargs.get("messages") or (args[0] if args else None)
133
208
  model = kwargs.get("model")
134
209
  input_repr = _safe_serialize(input_messages)
135
210
  tool_definitions = capture_tool_definitions(kwargs.get("tools"))
136
211
 
212
+ if kwargs.get("stream"):
213
+ # The parent is fixed at call time: the stream finalizes later, possibly inside an
214
+ # unrelated span (or none), and must not attach to whatever is active then.
215
+ parent = tracer.current_span
216
+
217
+ def on_stream_finish(collected: Dict[str, Any], error: Optional[str]) -> None:
218
+ global _warned_stream_usage
219
+ ttft = collected.get("time_to_first_token_ms")
220
+ call_metadata: Dict[str, Any] = {"streaming": True}
221
+ if ttft is not None:
222
+ call_metadata["timeToFirstTokenMs"] = ttft
223
+ if error is None and collected.get("input_tokens") is None and not _warned_stream_usage:
224
+ _warned_stream_usage = True
225
+ logger.warning(
226
+ "Streamed %s call carried no token usage - pass stream_options={\"include_usage\": True} "
227
+ "so traces (and cost) reflect streamed traffic.",
228
+ framework,
229
+ )
230
+ finish_llm_call(
231
+ tracer,
232
+ name=name,
233
+ framework=framework,
234
+ metadata=metadata,
235
+ call_metadata=call_metadata,
236
+ active_span=parent,
237
+ session_id=session_id,
238
+ start_t=start_t,
239
+ end_t=collected.get("end_t") or time.time(),
240
+ input_repr=input_repr,
241
+ output=collected.get("output"),
242
+ model=collected.get("model") or model,
243
+ input_tokens=collected.get("input_tokens"),
244
+ output_tokens=collected.get("output_tokens"),
245
+ cache_read_tokens=collected.get("cache_read_tokens"),
246
+ error=error,
247
+ tool_definitions=tool_definitions,
248
+ )
249
+
250
+ try:
251
+ result = original(*args, **kwargs)
252
+ except Exception as exc:
253
+ on_stream_finish({}, str(exc))
254
+ raise
255
+ return trace_stream(result, _ChatCompletionStreamAccumulator(start_t), on_stream_finish)
256
+
137
257
  def on_finish(response: Optional[Any], error: Optional[str]) -> None:
138
258
  end_t = time.time()
139
259
  output = None
@@ -12,6 +12,7 @@ from agentx.monitor.patterns import MonitorPatternBuilder, MonitorPatternClient
12
12
  from agentx.monitor.profile import MonitorProfileClient
13
13
  from agentx.monitor.review_queue import ReviewQueueClient, ReviewQueueItem
14
14
  from agentx.monitor.rules import MonitorRule, MonitorRulesClient
15
+ from agentx.monitor.alert_rules import AlertEvent, AlertRule, AlertRulesClient
15
16
  from agentx.monitor.scorers import AgentXScorersError, ScorersClient
16
17
  from agentx.monitor.scorer_groups import AgentXScorerGroupsError, ScorerGroup, ScorerGroupsClient
17
18
  from agentx.monitor.sessions import MonitorSessionClient
@@ -23,6 +24,9 @@ __all__ = [
23
24
  "AgentXMonitorError",
24
25
  "AgentXScorerGroupsError",
25
26
  "AgentXScorersError",
27
+ "AlertEvent",
28
+ "AlertRule",
29
+ "AlertRulesClient",
26
30
  "ImprovementGroupsClient",
27
31
  "JudgeScorer",
28
32
  "JudgeScorerBuilder",
@@ -0,0 +1,223 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import Any, Dict, List, Optional, TYPE_CHECKING
4
+
5
+ if TYPE_CHECKING:
6
+ from agentx.monitor.client import MonitorClient
7
+
8
+ ALERT_METRICS = ("failureRate", "toolFailureRate", "p95LatencyMs", "estimatedCostUsd", "judgeFailures", "traceCount")
9
+ ALERT_CHANNEL_KINDS = ("slack", "teams", "pagerduty", "email", "webhook")
10
+
11
+ # snake_case kwargs -> wire keys. The engine reads camelCase only (the wire convention); an
12
+ # unknown snake_case key would be silently ignored, so update() refuses it instead.
13
+ _ALIASES = {
14
+ "window_minutes": "windowMinutes",
15
+ "agent_id": "agentId",
16
+ "cooldown_minutes": "cooldownMinutes",
17
+ }
18
+
19
+
20
+ def _validate(
21
+ metric: Optional[str] = None, operator: Optional[str] = None, channels: Optional[List[Dict[str, str]]] = None
22
+ ) -> None:
23
+ """Local checks for the fields the engine would otherwise 400 on - ``None`` means "not given"
24
+ (an update that leaves the field alone)."""
25
+ if metric is not None and metric not in ALERT_METRICS:
26
+ raise ValueError(f"metric must be one of {ALERT_METRICS}, got {metric!r}")
27
+ if operator is not None and operator not in ("gt", "lt"):
28
+ raise ValueError(f"operator must be 'gt' or 'lt', got {operator!r}")
29
+ if channels is not None:
30
+ for channel in channels:
31
+ kind = channel.get("kind") if isinstance(channel, dict) else None
32
+ if kind not in ALERT_CHANNEL_KINDS:
33
+ raise ValueError(f"channel kind must be one of {ALERT_CHANNEL_KINDS}, got {kind!r}")
34
+
35
+
36
+ class AlertRule(dict):
37
+ """Wire object for one KPI alert rule (dict subclass so unknown fields round-trip)."""
38
+
39
+ @property
40
+ def id(self) -> str:
41
+ return self["_id"]
42
+
43
+ @property
44
+ def enabled(self) -> bool:
45
+ return bool(self.get("enabled"))
46
+
47
+ @property
48
+ def state(self) -> str:
49
+ """``"ok"`` or ``"firing"``."""
50
+ return str(self.get("state") or "ok")
51
+
52
+ @property
53
+ def last_value(self) -> Optional[float]:
54
+ return self.get("lastValue")
55
+
56
+ @property
57
+ def fired_count(self) -> int:
58
+ return int(self.get("firedCount") or 0)
59
+
60
+
61
+ class AlertEvent(dict):
62
+ """One row of a rule's notification history: ``kind`` is ``triggered`` / ``repeat`` /
63
+ ``resolved`` / ``test``; ``deliveries`` lists each channel's outcome."""
64
+
65
+ @property
66
+ def kind(self) -> str:
67
+ return str(self.get("kind") or "")
68
+
69
+ @property
70
+ def delivered(self) -> bool:
71
+ deliveries = self.get("deliveries") or []
72
+ return bool(deliveries) and all(isinstance(d, dict) and bool(d.get("ok")) for d in deliveries)
73
+
74
+
75
+ def slack(url: str) -> Dict[str, str]:
76
+ """A Slack incoming-webhook channel."""
77
+ return {"kind": "slack", "target": url}
78
+
79
+
80
+ def teams(url: str) -> Dict[str, str]:
81
+ """A Microsoft Teams incoming-webhook (or Workflows) channel."""
82
+ return {"kind": "teams", "target": url}
83
+
84
+
85
+ def pagerduty(routing_key: str) -> Dict[str, str]:
86
+ """A PagerDuty Events API v2 integration - ``routing_key`` is the integration key."""
87
+ return {"kind": "pagerduty", "target": routing_key}
88
+
89
+
90
+ def email(address: str) -> Dict[str, str]:
91
+ """An email recipient (needs a mailer configured on the engine)."""
92
+ return {"kind": "email", "target": address}
93
+
94
+
95
+ def webhook(url: str) -> Dict[str, str]:
96
+ """A generic JSON webhook receiving the full structured notification."""
97
+ return {"kind": "webhook", "target": url}
98
+
99
+
100
+ class AlertRulesClient:
101
+ """Surfaced as ``client.monitor.alert_rules``: KPI alert rules.
102
+
103
+ An alert rule watches an AGGREGATE over a sliding window - one of
104
+ ``failureRate``, ``toolFailureRate``, ``p95LatencyMs``, ``estimatedCostUsd``,
105
+ ``judgeFailures``, ``traceCount`` - and pages typed channels (Slack, Teams,
106
+ PagerDuty, email, generic webhook) when it crosses a threshold. The engine
107
+ evaluates every enabled rule once a minute with an Alertmanager-style
108
+ lifecycle: one ``triggered`` notification when the rule starts breaching, a
109
+ ``repeat`` every ``cooldown_minutes`` while it keeps breaching, and a
110
+ ``resolved`` notification when it recovers. This is distinct from a scorer's
111
+ per-verdict alert threshold and from an automation rule's per-trace routing.
112
+
113
+ Example::
114
+
115
+ from agentx.monitor.alert_rules import slack, pagerduty
116
+
117
+ rule = client.monitor.alert_rules.create(
118
+ "Failure rate above 10%",
119
+ metric="failureRate", operator="gt", threshold=0.10, window_minutes=15,
120
+ severity="high",
121
+ channels=[slack("https://hooks.slack.com/services/..."), pagerduty("R0123...")],
122
+ )
123
+ client.monitor.alert_rules.test(rule.id) # sends a TEST page to every channel
124
+ """
125
+
126
+ def __init__(self, client: "MonitorClient"):
127
+ self._client = client
128
+
129
+ def _request(self, method: str, path: str, **kwargs: Any) -> Any:
130
+ return self._client._request(method, path, base=self._client._api_root(), **kwargs)
131
+
132
+ def list(self) -> List[AlertRule]:
133
+ data = self._request("GET", "/agent-monitoring/alert-rules")
134
+ return [AlertRule(r) for r in data.get("rules", [])]
135
+
136
+ def get(self, rule_id: str) -> AlertRule:
137
+ data = self._request("GET", f"/agent-monitoring/alert-rules/{rule_id}")
138
+ return AlertRule(data.get("rule", data))
139
+
140
+ def create(
141
+ self,
142
+ name: str,
143
+ *,
144
+ metric: str,
145
+ operator: str,
146
+ threshold: float,
147
+ window_minutes: int,
148
+ channels: List[Dict[str, str]],
149
+ agent_id: Optional[str] = None,
150
+ severity: str = "high",
151
+ cooldown_minutes: int = 60,
152
+ enabled: bool = True,
153
+ ) -> AlertRule:
154
+ """Create a rule. ``operator`` is ``"gt"`` (above) or ``"lt"`` (below); rates are
155
+ fractions (``0.10`` = 10%), latency is milliseconds, cost is USD. ``channels`` takes the
156
+ dicts the module-level helpers build (``slack(url)``, ``pagerduty(key)``, ...)."""
157
+ _validate(metric=metric, operator=operator, channels=channels)
158
+ payload: Dict[str, Any] = {
159
+ "name": name,
160
+ "metric": metric,
161
+ "operator": operator,
162
+ "threshold": threshold,
163
+ "windowMinutes": window_minutes,
164
+ "channels": channels,
165
+ "severity": severity,
166
+ "cooldownMinutes": cooldown_minutes,
167
+ "enabled": enabled,
168
+ }
169
+ if agent_id is not None:
170
+ payload["agentId"] = agent_id
171
+ # Server-side write: a timeout retry would create a duplicate rule that pages twice on
172
+ # every incident - no transport retry (same posture as rules.create).
173
+ data = self._request("POST", "/agent-monitoring/alert-rules", json=payload, retry=False)
174
+ return AlertRule(data.get("rule", data))
175
+
176
+ def update(self, rule_id: str, **fields: Any) -> AlertRule:
177
+ """Sparse update; snake_case kwargs are mapped to the wire. Changing the metric,
178
+ operator, threshold, window, or agent resets the rule's firing state, and a rule that
179
+ was firing sends its channels a final ``resolved`` notification first. The same local
180
+ checks as ``create`` apply to whichever of ``metric``, ``operator``, ``channels`` are
181
+ given."""
182
+ payload: Dict[str, Any] = {}
183
+ for key, value in fields.items():
184
+ wire_key = _ALIASES.get(key, key)
185
+ if "_" in wire_key:
186
+ raise ValueError(
187
+ f"Unknown alert rule field {key!r} - the engine reads camelCase keys and would "
188
+ "silently ignore this (see AlertRulesClient.create for the field names)."
189
+ )
190
+ payload[wire_key] = value
191
+ _validate(metric=payload.get("metric"), operator=payload.get("operator"), channels=payload.get("channels"))
192
+ data = self._request("PUT", f"/agent-monitoring/alert-rules/{rule_id}", json=payload)
193
+ return AlertRule(data.get("rule", data))
194
+
195
+ def delete(self, rule_id: str) -> None:
196
+ """Deletes the rule and its history. A rule that is firing sends its channels a final
197
+ ``resolved`` notification (closing the PagerDuty incident it opened) before it goes."""
198
+ self._request("DELETE", f"/agent-monitoring/alert-rules/{rule_id}", retry=False)
199
+
200
+ def events(self, rule_id: str, limit: int = 50) -> List[AlertEvent]:
201
+ """The rule's notification history, newest first, with per-channel delivery results."""
202
+ data = self._request("GET", f"/agent-monitoring/alert-rules/{rule_id}/events", params={"limit": limit})
203
+ return [AlertEvent(e) for e in data.get("events", [])]
204
+
205
+ def test(self, rule_id: str) -> AlertEvent:
206
+ """Send a TEST notification to the rule's channels with the metric's live value and
207
+ return the recorded event - ``event.delivered`` says whether every channel accepted it.
208
+ Never changes the rule's firing state."""
209
+ data = self._request("POST", f"/agent-monitoring/alert-rules/{rule_id}/test", json={}, retry=False)
210
+ return AlertEvent(data.get("event", data))
211
+
212
+ def preview(self, metric: str, window_minutes: int, agent_id: Optional[str] = None) -> Dict[str, Any]:
213
+ """What ``metric`` reads right now over the last ``window_minutes`` - the same
214
+ computation the sweep runs. Returns ``{"value": float | None, "valueLabel": str, ...}``;
215
+ ``value`` is ``None`` when the window has no data for a rate metric."""
216
+ payload: Dict[str, Any] = {"metric": metric, "windowMinutes": window_minutes}
217
+ if agent_id is not None:
218
+ payload["agentId"] = agent_id
219
+ return self._request("POST", "/agent-monitoring/alert-rules/preview", json=payload)
220
+
221
+ def run_sweep(self) -> Dict[str, Any]:
222
+ """Evaluate this project's rules now instead of waiting for the next minute tick."""
223
+ return self._request("POST", "/agent-monitoring/alert-rules/sweep/run", json={}, retry=False)
agentx/monitor/client.py CHANGED
@@ -139,6 +139,11 @@ class MonitorClient:
139
139
 
140
140
  # Automation rules: route matching traffic into review / a dataset / a webhook.
141
141
  self.rules = MonitorRulesClient(self)
142
+ from agentx.monitor.alert_rules import AlertRulesClient
143
+
144
+ # KPI alert rules: threshold pages (Slack/Teams/PagerDuty/email/webhook) on failure
145
+ # rate, p95 latency, spend, judge failures, and traffic volume over a window.
146
+ self.alert_rules = AlertRulesClient(self)
142
147
  from agentx.monitor.scorers import ScorersClient
143
148
  # Scorers-catalog administration as code: template enable/disable, code/external scorer
144
149
  # CRUD and dry runs - full parity with the dashboard's Scorers page (P1.3).
agentx/tracing/tracer.py CHANGED
@@ -165,6 +165,9 @@ class _TraceSpan:
165
165
  self.tool_calls: list = []
166
166
 
167
167
  self._start: Optional[float] = None
168
+ # Set by callers that know when the work actually ended (a streamed LLM call's last
169
+ # chunk) so __exit__ does not measure to "now" - see finish_llm_call's root path.
170
+ self._end_override: Optional[float] = None
168
171
  self._error: Optional[str] = None
169
172
 
170
173
  self._captured_model: Optional[str] = None
@@ -212,7 +215,8 @@ class _TraceSpan:
212
215
 
213
216
  def __exit__(self, exc_type, exc_val, tb):
214
217
  self._tracer._pop_active_span(self)
215
- latency_ms = int((time.time() - self._start) * 1000) if self._start else None
218
+ ended_at = self._end_override if self._end_override is not None else time.time()
219
+ latency_ms = int((ended_at - self._start) * 1000) if self._start else None
216
220
  if exc_val is not None and self._error is None:
217
221
  self._error = str(exc_val)
218
222
 
@@ -297,6 +301,7 @@ class _TraceSpan:
297
301
  output_tokens: Optional[int] = None,
298
302
  cache_read_tokens: Optional[int] = None,
299
303
  cache_write_tokens: Optional[int] = None,
304
+ metadata: Optional[Dict[str, Any]] = None,
300
305
  ) -> None:
301
306
  """Record one LLM-call child span (e.g. one patched Anthropic call) under this span -
302
307
  name left unset so _merge_child_run auto-numbers it "LLM Call N". ``framework`` lets the
@@ -315,6 +320,9 @@ class _TraceSpan:
315
320
  "outputTokenSize": output_tokens,
316
321
  "cacheReadTokenSize": cache_read_tokens,
317
322
  "cacheWriteTokenSize": cache_write_tokens,
323
+ # Per-call facts that belong on the child row (a streamed call's time to first
324
+ # token), not on the parent trace's metadata.
325
+ "metadata": metadata,
318
326
  }],
319
327
  input=input,
320
328
  output=output,
@@ -479,6 +487,7 @@ class _TraceSpan:
479
487
  output_tokens=step.get("outputTokenSize"),
480
488
  cache_read_tokens=step.get("cacheReadTokenSize"),
481
489
  cache_write_tokens=step.get("cacheWriteTokenSize"),
490
+ metadata=step.get("metadata") or None,
482
491
  # Stated, so a step named anything other than "LLM Call N" still classifies -
483
492
  # the backend's name regex was the only thing holding this together. Steps
484
493
  # may state their own kind (crewai.py's task steps carry "agent"); the
agentx/version.py CHANGED
@@ -1,7 +1,7 @@
1
- VERSION = "0.8.26"
1
+ VERSION = "0.8.28"
2
2
 
3
3
  # The AgentX-trace-eval release this SDK version is tested against - what `agentx-trace-eval`
4
4
  # installs and converges to (see agentx/cli.py). Bump together with VERSION when releasing, so
5
5
  # every published SDK names a known-good engine+dashboard pair. Users can override with
6
6
  # AGENTX_TRACE_EVAL_VERSION=<tag|latest>.
7
- ENGINE_VERSION = "v0.3.29"
7
+ ENGINE_VERSION = "v0.3.31"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentx-python
3
- Version: 0.8.26
3
+ Version: 0.8.28
4
4
  Summary: Official Python SDK for AgentX (https://www.agentx.so/)
5
5
  Home-page: https://github.com/AgentX-ai/AgentX-python
6
6
  Author: Robin Wang and AgentX Team
@@ -260,6 +260,12 @@ extra:
260
260
  | LlamaIndex | `pip install "agentx-python[llamaindex]"` | `AgentXLlamaIndexHandler` |
261
261
  | AutoGen | `pip install "agentx-python[autogen]"` | `AgentXAutoGenObserver` |
262
262
 
263
+ Raw-client patches (`patch_openai_client`, `patch_nim_client`, `patch_anthropic_client`) trace
264
+ streaming calls too: the returned stream is a transparent proxy that assembles the reply from the
265
+ chunks you consume, with latency measured to the last chunk and the time to first token in the
266
+ trace metadata. OpenAI-compatible endpoints only send token usage on streams when you pass
267
+ `stream_options={"include_usage": True}`.
268
+
263
269
  > **Warning: pick ONE instrumentation layer per LLM call.** Do not combine
264
270
  > `AgentXCallbackHandler` (or any framework integration) with a patched provider client
265
271
  > (`patch_openai_client`, `patch_anthropic_client`, `patch_genai_client`) on the same code
@@ -10,7 +10,7 @@ agentx/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
10
10
  agentx/testing.py,sha256=0shZEid_vhJgBegpO_loUq6APYDxQrUgD3jvcRdIDV8,7372
11
11
  agentx/traces.py,sha256=Jh07n3GLS1SGRsDoLDTUaAbuMEdvjqxCCMA9gUcSZ0M,2331
12
12
  agentx/util.py,sha256=ivCuFQ5AGUw9fR9fJy5HrSp3lLwwngsDwppnvyZpq9E,1471
13
- agentx/version.py,sha256=mHuhKZSPSp8Xp5Pq7q93wgdcMtBUM_OBt-nWG4Qoh6g,366
13
+ agentx/version.py,sha256=GG8-5mWn22akjdeFMAvH6YuyN72ciRGt8EIqmtQjbdw,366
14
14
  agentx/evaluations/__init__.py,sha256=Erv7RGFlRxGTG4rVb2uHhCqLLEX_iAWKqkZKzK6CumE,262
15
15
  agentx/evaluations/_term.py,sha256=WFpiNzdgDBeJJ-Gg-6X7TwwxllobuE4OqFUTuDQvS3Y,2529
16
16
  agentx/evaluations/client.py,sha256=7c35tS8zM3fuvFfVDQk5etTTEFiIykmPjuJxkuFyFTY,38013
@@ -28,8 +28,8 @@ agentx/evaluations/adapters/http_endpoint.py,sha256=-Gika9lBcXjtbnyB-uNR6nYEVJhW
28
28
  agentx/evaluations/adapters/precomputed.py,sha256=vxnQELmpiU2NkQATfiVjG7v1ZmmkrEA70Y11ijfYtHw,1208
29
29
  agentx/evaluations/adapters/raw.py,sha256=FkDq_mdf21-xt5EHnGyIGcS6VZxkEkFcmM9VTd-T5sA,1130
30
30
  agentx/integrations/__init__.py,sha256=p-YHLIudYxQj1bBj77NiVPsP0VQHemX5E59phkI3Y-Q,711
31
- agentx/integrations/_traced_call.py,sha256=OuFmSLw5vjI54gpdte7-zOt7gld7faQmvhuoh_3OPlo,6861
32
- agentx/integrations/anthropic.py,sha256=z6o9cC9rBzxxbF1UBiMr-kjEgDUO-BCuPOFoqGN3qF4,10451
31
+ agentx/integrations/_traced_call.py,sha256=N-Yk1Bos2rEdlE6h9OvJYqK54FPLqUJD4dN7vOwhnK0,17602
32
+ agentx/integrations/anthropic.py,sha256=44KEiTv830xjMB2rNyXe92Y-Ha3laxAyATR0qubiQEo,18117
33
33
  agentx/integrations/autogen.py,sha256=V1kvTjxQqOG9KsYd4dztaRYH7zTknW0_fqzKjIQG5dk,9507
34
34
  agentx/integrations/crewai.py,sha256=SCObRsrS40P8jOGoI_T5oW4IYuoBW5WjBNuc7zUBU70,14346
35
35
  agentx/integrations/databricks.py,sha256=vKXnur-LWMTzctFuIKte2N826sydXVlpi1qFXo2dfio,18044
@@ -39,13 +39,14 @@ agentx/integrations/langchain.py,sha256=YDeEFT2irhoEvydwt3QjbmTpk_AZZbYPS9euOR18
39
39
  agentx/integrations/litellm.py,sha256=eW8iCaNq0feRbZma8iOvmJ28WiAO7ZaRmlpPjHAtcvo,7072
40
40
  agentx/integrations/llamaindex.py,sha256=ruzMiC2TC3Xhub4xoXKPHjsvzV1sQ3QqhjkAdMxUUi4,18001
41
41
  agentx/integrations/moveworks.py,sha256=IyBswLE5LwMSp1VbIvG5izYV5BvXmv4atZelnrDn8SE,25670
42
- agentx/integrations/nvidia_nim.py,sha256=bvS6ebEnQZT1DVlIjQQ2RKLI4uPPTdlMKINgDwj0Z-o,2961
43
- agentx/integrations/openai.py,sha256=edJL-HXTO7w-vvIZGQChMu0AnPCY5z0duplIi8z2gXE,6936
42
+ agentx/integrations/nvidia_nim.py,sha256=SVt_WSTsDvyxrCwbioeQ-NZWQL9PI8zXl9Y7ObXUVsU,3165
43
+ agentx/integrations/openai.py,sha256=WWD2SgqZa-zdz54rrOP739cj5vuImbkE6rg-tTC7Slw,12389
44
44
  agentx/integrations/openai_agents.py,sha256=KSOQ55eRPFxxkqm7p4Dsz6HcJ_9uUEwlyywPI8dwbvY,14852
45
- agentx/monitor/__init__.py,sha256=Rkk-Xv1ymXwDFbx8fetmcOzXfSU1bvsazI_-x5Ati5g,1715
45
+ agentx/monitor/__init__.py,sha256=3KsLb0wedvh16wa8W2G3Z2bZ-LPxe5xmw2wAYmqtfZA,1853
46
46
  agentx/monitor/_transport.py,sha256=ge0pmdvFdUtCJRkTLvAgjuTQVA8YOCYGQNoyRuselRk,2077
47
47
  agentx/monitor/agents.py,sha256=KhsHVhkqa6V0e0lPcuo04fh44_qSaJOaGcD6U6qXSFo,1622
48
- agentx/monitor/client.py,sha256=JoKp3AmFtQ4UDOsNH5nWEkTjvQtBrrihObGaPVMtBYo,27270
48
+ agentx/monitor/alert_rules.py,sha256=KIZoxyjPG4snnwmO3h4GEbKtlWULmcwGqNMLsdweXew,9871
49
+ agentx/monitor/client.py,sha256=oZctmh77Ki5WnZI6vQaj_tDlmdP6XtkDYt3MKhnvGsU,27563
49
50
  agentx/monitor/improvement_groups.py,sha256=MrXexaBLPNzZHrqZjiCPfmwa-DSr-RHuFdP73SKrMJM,6143
50
51
  agentx/monitor/judge_scorers.py,sha256=Xyoafo9ljeq5RpLe5HxvsL0Y0FebFQ-2CxYz8-E1Wio,21025
51
52
  agentx/monitor/models.py,sha256=mNBQDXdVthGYHUyEeRgpFdKVOC9uMtdUgBlLfep1E14,9298
@@ -67,10 +68,10 @@ agentx/tracing/ci_types.py,sha256=b-W2LowRhNBVUdaoMPgd4bnKQy9AxhfiikftRwGPAqU,17
67
68
  agentx/tracing/eval_scope.py,sha256=ElMbPxpqpQBVuaUnHNlw9RIQu71oyOpd0yR55bDH8eI,2144
68
69
  agentx/tracing/framework_detect.py,sha256=uV4O7Th-4_2UkdooAyaWCOIA0jwLJbblFQ_ucFDjeJM,2638
69
70
  agentx/tracing/ingest_client.py,sha256=teukTPckFWPg5RhpbGbVDBejBzPIPXUO3_DJcuVh2uc,22660
70
- agentx/tracing/tracer.py,sha256=NS0bs5HbDDK3l0Z95KVKV3xCNJT7LINS0EoPFM-wzIk,68543
71
- agentx_python-0.8.26.dist-info/licenses/LICENSE,sha256=gZVsM-nLsE8vlaY6NXXsVoo6IlCClxkToAWmhgT3y_s,10762
72
- agentx_python-0.8.26.dist-info/METADATA,sha256=JH0SY8e_dBGzmK92xd_Cas62qpPweH0BoBLbDmE7xPA,22986
73
- agentx_python-0.8.26.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
74
- agentx_python-0.8.26.dist-info/entry_points.txt,sha256=rQqF1JTY3T1yfviBU1r8PU-5mBdfOk6P4bZ1L4BskIM,172
75
- agentx_python-0.8.26.dist-info/top_level.txt,sha256=s-q-HB9Gb_QdrZNacSeQyF_c25gQooMy7DlxzgLOHPk,7
76
- agentx_python-0.8.26.dist-info/RECORD,,
71
+ agentx/tracing/tracer.py,sha256=N7rIobHM9KdPU9mqBLVqgk_1mqtCNjBD57uAtXn9DO0,69167
72
+ agentx_python-0.8.28.dist-info/licenses/LICENSE,sha256=gZVsM-nLsE8vlaY6NXXsVoo6IlCClxkToAWmhgT3y_s,10762
73
+ agentx_python-0.8.28.dist-info/METADATA,sha256=9sFd2gVCfCWgweOoBcTHqM4IUpfHipe0qxYBEiLU4E4,23408
74
+ agentx_python-0.8.28.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
75
+ agentx_python-0.8.28.dist-info/entry_points.txt,sha256=rQqF1JTY3T1yfviBU1r8PU-5mBdfOk6P4bZ1L4BskIM,172
76
+ agentx_python-0.8.28.dist-info/top_level.txt,sha256=s-q-HB9Gb_QdrZNacSeQyF_c25gQooMy7DlxzgLOHPk,7
77
+ agentx_python-0.8.28.dist-info/RECORD,,