agentx-python 0.8.18__py3-none-any.whl → 0.8.20__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
agentx/__init__.py CHANGED
@@ -12,11 +12,10 @@ from agentx.exceptions import (
12
12
  CIGateFailure,
13
13
  )
14
14
 
15
- logging.basicConfig(
16
- level=logging.INFO,
17
- format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
18
- datefmt="%Y-%m-%d %H:%M:%S %Z",
19
- )
15
+ # Library logging hygiene: a library must never call logging.basicConfig - it hijacks the
16
+ # host application's root logger (format AND level) and turns the app's own later basicConfig
17
+ # into a no-op. Consumers opt into our logs with logging.getLogger("agentx").setLevel(...).
18
+ logging.getLogger("agentx").addHandler(logging.NullHandler())
20
19
 
21
20
  __all__ = [
22
21
  "AgentX",
@@ -348,7 +348,9 @@ class EvaluationsClient:
348
348
  payload["scorerGroupId"] = scorer_group_id
349
349
  if split:
350
350
  payload["split"] = split
351
- data = self._request("POST", "/runs", json=self._with_workspace(payload))
351
+ # Server-side write: a timeout after the run row was created would be
352
+ # retried into a duplicate run, so no transport retry.
353
+ data = self._request("POST", "/runs", json=self._with_workspace(payload), retry=False)
352
354
  return EvaluationRun(**data)
353
355
 
354
356
  def append_results(
@@ -432,8 +434,12 @@ class EvaluationsClient:
432
434
 
433
435
  if not self._analysis_on_dashboard_router:
434
436
  try:
437
+ # The self-host route runs the analysis SYNCHRONOUSLY (engine
438
+ # routes/evaluations.ts) - a short timeout with retries re-billed the whole
439
+ # multi-judge analysis up to 4x while the first was still running. Full
440
+ # analysis timeout, no transport retry.
435
441
  return self._request(
436
- "POST", f"/runs/{run_id}/analyze", json=payload, timeout=30
442
+ "POST", f"/runs/{run_id}/analyze", json=payload, timeout=1800, retry=False
437
443
  )
438
444
  except AgentXEvaluationsError as exc:
439
445
  if not self._note_missing_analysis_route(exc, "analyze"):
@@ -554,13 +560,17 @@ class EvaluationsClient:
554
560
  if isinstance(dataset_id, dict): # populated reference, not a bare id
555
561
  dataset_id = dataset_id.get("_id") or dataset_id.get("id")
556
562
 
557
- return Report(
558
- runId=run_id,
559
- datasetId=dataset_id or "",
560
- status=envelope.get("status") or "completed",
561
- statistics=envelope.get("statistics"),
563
+ # Built as one merged dict (explicit keys last, so they win) - passing the
564
+ # explicit keys as keyword arguments alongside **body raises "got multiple
565
+ # values" whenever the analysis body itself carries runId/datasetId/status/
566
+ # statistics.
567
+ return Report(**{
562
568
  **body,
563
- )
569
+ "runId": run_id,
570
+ "datasetId": dataset_id or "",
571
+ "status": envelope.get("status") or "completed",
572
+ "statistics": envelope.get("statistics"),
573
+ })
564
574
 
565
575
  # ------------------------------------------------------------------
566
576
  # Prompt improvement loop (examples -> propose -> publish). These ride the engine's
@@ -589,8 +599,11 @@ class EvaluationsClient:
589
599
  payload["reasoning"] = reasoning
590
600
  if based_on_version is not None:
591
601
  payload["basedOnVersion"] = based_on_version
602
+ # Server-side write: a timeout after the version was stored would be
603
+ # retried into a duplicate version, so no transport retry.
592
604
  return self._request(
593
- "POST", f"/evaluate/prompts/{prompt_id}/versions", base=self._api_root, json=payload
605
+ "POST", f"/evaluate/prompts/{prompt_id}/versions", base=self._api_root, json=payload,
606
+ retry=False,
594
607
  )
595
608
 
596
609
  # ------------------------------------------------------------------
@@ -628,7 +641,7 @@ class EvaluationsClient:
628
641
  payload["judgeModel"] = judge_model
629
642
  if both_orders:
630
643
  payload["bothOrders"] = True
631
- response = self._request("POST", "/evaluate/runs/pairwise", json=payload, base=self._api_root)
644
+ response = self._request("POST", "/evaluate/runs/pairwise", json=payload, base=self._api_root, timeout=900, retry=False,)
632
645
  return PairwiseComparison(**response["comparison"])
633
646
 
634
647
  def get_pairwise(self, batch_id: str) -> PairwiseComparison:
@@ -662,7 +675,8 @@ class EvaluationsClient:
662
675
  payload: dict = {"name": name, "definition": definition}
663
676
  if description is not None:
664
677
  payload["description"] = description
665
- return self._request("POST", "/evaluate/tool-schemas", base=self._api_root, json=payload)
678
+ # Server-side write - no transport retry (see init_run's comment).
679
+ return self._request("POST", "/evaluate/tool-schemas", base=self._api_root, json=payload, retry=False)
666
680
 
667
681
  def get_tool_schema_examples(self, tool_schema_id: str, window: Optional[str] = None) -> dict:
668
682
  params = {"window": window} if window else None
@@ -686,8 +700,10 @@ class EvaluationsClient:
686
700
  payload["reasoning"] = reasoning
687
701
  if based_on_version is not None:
688
702
  payload["basedOnVersion"] = based_on_version
703
+ # Server-side write - no transport retry (see init_run's comment).
689
704
  return self._request(
690
- "POST", f"/evaluate/tool-schemas/{tool_schema_id}/versions", base=self._api_root, json=payload
705
+ "POST", f"/evaluate/tool-schemas/{tool_schema_id}/versions", base=self._api_root, json=payload,
706
+ retry=False,
691
707
  )
692
708
 
693
709
  # ------------------------------------------------------------------
@@ -16,7 +16,13 @@ _REQUIRED_CSV_COLS = {"query"}
16
16
 
17
17
 
18
18
  class DatasetBuilder:
19
- """Fluent builder for creating a Custom Agent Evaluations dataset."""
19
+ """Fluent builder for creating a Custom Agent Evaluations dataset.
20
+
21
+ ``judge_prompt``/``judge_model`` are LLM-as-judge overrides for the dataset's grading
22
+ config. NOTE (self-host): the engine's dataset-create route currently ignores both -
23
+ set them on a judge scorer / the evaluation settings instead. ``sovereignty_models``
24
+ is accepted on the wire but not acted on by the self-host engine.
25
+ """
20
26
 
21
27
  def __init__(
22
28
  self,
@@ -50,6 +56,8 @@ class DatasetBuilder:
50
56
  # LLM-as-judge overrides for this dataset's own grading config. Omit either to keep the
51
57
  # server default (raw prompt template / gpt-5.6-luna, see EVALUATIONS.md). judge_model
52
58
  # must be one of client.evaluations.list_models() (OpenAI or Anthropic).
59
+ # Self-host: the dataset-create route currently IGNORES judgePrompt/judgeModel - set
60
+ # them on a judge scorer / the evaluation settings instead (see class docstring).
53
61
  if judge_prompt is not None:
54
62
  self._payload["judgePrompt"] = judge_prompt
55
63
  if judge_model is not None:
@@ -84,7 +92,8 @@ class DatasetBuilder:
84
92
  if rouge_score:
85
93
  self._payload["rougeScore"] = {"enabled": True}
86
94
  # Sovereignty & Portability - the models to compare on this dataset (use
87
- # client.evaluations.list_models() to discover valid ids).
95
+ # client.evaluations.list_models() to discover valid ids). Self-host: accepted on
96
+ # the wire but not acted on by the engine (see class docstring).
88
97
  if sovereignty_models:
89
98
  self._payload["sovereigntyIndex"] = {
90
99
  "enabled": True,
@@ -354,10 +363,13 @@ class DatasetClient:
354
363
  "acceptanceCriteria",
355
364
  "rejectionCriteria",
356
365
  "evaluationCriteria",
366
+ "judgePrompt",
367
+ "judgeModel",
357
368
  "vectorSimilarity",
358
369
  "jaccardSimilarity",
359
370
  "bleuScore",
360
371
  "rougeScore",
372
+ "sovereigntyIndex",
361
373
  "codeScorers",
362
374
  ):
363
375
  if wire.get(key) is not None:
@@ -1,5 +1,7 @@
1
1
  from __future__ import annotations
2
2
 
3
+ from typing import Optional
4
+
3
5
  from agentx.evaluations.models import Report
4
6
  from agentx.evaluations._term import (
5
7
  bold,
@@ -21,14 +23,14 @@ _RATING_ICONS = {"high": "●", "medium": "◑", "low": "○"}
21
23
  _PRI_COLORS = {"high": red, "medium": yellow, "low": dim}
22
24
 
23
25
 
24
- def _rating_badge(rating: str | None) -> str:
26
+ def _rating_badge(rating: Optional[str]) -> str:
25
27
  icon = _RATING_ICONS.get(rating or "", "·")
26
28
  color = _RATING_COLORS.get(rating or "", dim)
27
29
  label = (rating or "").upper()
28
30
  return color(f"{icon} {label}") if label else dim(icon)
29
31
 
30
32
 
31
- def _section(title: str, rating: str | None = None) -> None:
33
+ def _section(title: str, rating: Optional[str] = None) -> None:
32
34
  badge = f" {_rating_badge(rating)}" if rating else ""
33
35
  print(f"\n{bold(title)}{badge}")
34
36
  print(dim(_THIN))
@@ -3,6 +3,8 @@ from __future__ import annotations
3
3
  import logging
4
4
  import os
5
5
  import time
6
+
7
+ import requests
6
8
  import uuid
7
9
  from typing import Any, Callable, Dict, Iterator, List, Optional, Set, Union
8
10
 
@@ -304,6 +306,12 @@ class EvaluationRunContext:
304
306
  resp.failed_validation,
305
307
  )
306
308
  return
309
+ except requests.Timeout as exc:
310
+ # A read timeout means the engine may STILL be scoring this batch - a
311
+ # retry re-POSTs it and double-bills every judge call (idempotency keys
312
+ # protect rows already inserted, not judge work mid-flight). Fail loud.
313
+ last_exc = exc
314
+ break
307
315
  except Exception as exc:
308
316
  last_exc = exc
309
317
  if attempt == 1:
@@ -274,7 +274,7 @@ def _patch_stream(
274
274
  return iter(ctx)
275
275
 
276
276
  def __aiter__(self_inner):
277
- return aiter(ctx)
277
+ return ctx.__aiter__() # aiter() builtin is 3.10+; python_requires is >=3.9
278
278
 
279
279
  def __getattr__(self_inner, item):
280
280
  return getattr(ctx, item)
@@ -167,6 +167,9 @@ class AgentXAutoGenObserver:
167
167
  "end_time": end_t,
168
168
  "input": pending["input"] if pending else None,
169
169
  "output": f"ERROR: {output}" if is_error else (str(output) if output is not None else None),
170
+ # The engine's failure test is success === false; without
171
+ # this a failed tool call would read as passing.
172
+ "success": not is_error,
170
173
  })
171
174
  continue
172
175
 
@@ -127,6 +127,16 @@ class AgentXCrewObserver:
127
127
  except ImportError:
128
128
  return task_timings, lambda: None
129
129
 
130
+ # Double-instrumentation guard (bus-keyed latch, the same idea as the
131
+ # other integrations' _agentx_patched flag): the event bus is a global
132
+ # singleton, so a notebook re-run or an overlapping kickoff that
133
+ # already has AgentX listeners registered would otherwise get a second
134
+ # set and duplicate every task span. When already attached, this
135
+ # kickoff just falls back to the evenly-divided timing approximation.
136
+ if getattr(crewai_event_bus, "_agentx_attached", False):
137
+ return task_timings, lambda: None
138
+ crewai_event_bus._agentx_attached = True
139
+
130
140
  def on_task_started(source: Any, event: Any) -> None:
131
141
  task_id = getattr(event, "task_id", None)
132
142
  if task_id is None:
@@ -154,9 +164,14 @@ class AgentXCrewObserver:
154
164
  crewai_event_bus.on(TaskFailedEvent)(on_task_failed)
155
165
 
156
166
  def unregister() -> None:
157
- crewai_event_bus.off(TaskStartedEvent, on_task_started)
158
- crewai_event_bus.off(TaskCompletedEvent, on_task_completed)
159
- crewai_event_bus.off(TaskFailedEvent, on_task_failed)
167
+ try:
168
+ crewai_event_bus.off(TaskStartedEvent, on_task_started)
169
+ crewai_event_bus.off(TaskCompletedEvent, on_task_completed)
170
+ crewai_event_bus.off(TaskFailedEvent, on_task_failed)
171
+ finally:
172
+ # Clear the latch even if .off() raises, so a later kickoff
173
+ # can re-attach instead of being locked out forever.
174
+ crewai_event_bus._agentx_attached = False
160
175
 
161
176
  return task_timings, unregister
162
177
 
@@ -339,6 +339,16 @@ class AgentXCallbackHandler(BaseCallbackHandler):
339
339
  self._retrieval_starts.pop(run_id, None)
340
340
  self._parents.pop(run_id, None)
341
341
 
342
+ # Pre-run retrieval steps waiting for a top-level chain that never came
343
+ # (e.g. retriever.invoke() called but agent.invoke() aborted before
344
+ # on_chain_start). Each step carries its own start_time, so drop the
345
+ # pre-cutoff ones just like the run_id-keyed structures above.
346
+ with self._state_lock:
347
+ if self._pending_retrieval_steps:
348
+ self._pending_retrieval_steps[:] = [
349
+ step for step in self._pending_retrieval_steps if step.get("start_time", 0) >= cutoff
350
+ ]
351
+
342
352
  # ------------------------------------------------------------------
343
353
  # Chain lifecycle
344
354
  # ------------------------------------------------------------------
@@ -359,8 +369,9 @@ class AgentXCallbackHandler(BaseCallbackHandler):
359
369
  self._prune_stale_entries()
360
370
  # Consume any retrieval steps that ran before this chain started
361
371
  # (pre-run RAG: retriever.invoke() called before agent.invoke())
362
- pending = self._pending_retrieval_steps[:]
363
- self._pending_retrieval_steps.clear()
372
+ with self._state_lock:
373
+ pending = self._pending_retrieval_steps[:]
374
+ self._pending_retrieval_steps.clear()
364
375
  self._runs[run_id] = {
365
376
  "start": time.time(),
366
377
  "input": _extract_input(inputs),
@@ -670,7 +681,7 @@ class AgentXCallbackHandler(BaseCallbackHandler):
670
681
  "input": _extract_llm_input(prompts=prompts, messages=messages),
671
682
  }
672
683
  top = self._find_top_ancestor(parent_run_id)
673
- if top and not self._runs[top].get("model") and model:
684
+ if top and top in self._runs and not self._runs[top].get("model") and model:
674
685
  self._runs[top]["model"] = model
675
686
 
676
687
  def on_llm_start(
@@ -882,15 +893,16 @@ class AgentXCallbackHandler(BaseCallbackHandler):
882
893
  step["output"] = "\n\n---\n\n".join(contents)
883
894
 
884
895
  top = self._find_top_ancestor(parent_run_id)
885
- if top and top in self._runs:
886
- # Retriever ran inside an active chain - attach directly
887
- retrievals = self._runs[top]["retrieval_steps"]
888
- step["name"] = f"Retrieval {len(retrievals) + 1}"
889
- retrievals.append(step)
890
- else:
891
- # Retriever ran before the chain started (pre-run RAG pattern)
892
- step["name"] = f"Retrieval {len(self._pending_retrieval_steps) + 1}"
893
- self._pending_retrieval_steps.append(step)
896
+ with self._state_lock:
897
+ if top and top in self._runs:
898
+ # Retriever ran inside an active chain - attach directly
899
+ retrievals = self._runs[top]["retrieval_steps"]
900
+ step["name"] = f"Retrieval {len(retrievals) + 1}"
901
+ retrievals.append(step)
902
+ else:
903
+ # Retriever ran before the chain started (pre-run RAG pattern)
904
+ step["name"] = f"Retrieval {len(self._pending_retrieval_steps) + 1}"
905
+ self._pending_retrieval_steps.append(step)
894
906
 
895
907
  def on_retriever_error(
896
908
  self,
@@ -918,15 +930,16 @@ class AgentXCallbackHandler(BaseCallbackHandler):
918
930
  step["query"] = query
919
931
 
920
932
  top = self._find_top_ancestor(parent_run_id)
921
- if top and top in self._runs:
922
- # Retriever ran inside an active chain - attach directly
923
- retrievals = self._runs[top]["retrieval_steps"]
924
- step["name"] = f"Retrieval {len(retrievals) + 1}"
925
- retrievals.append(step)
926
- else:
927
- # Retriever ran before the chain started (pre-run RAG pattern)
928
- step["name"] = f"Retrieval {len(self._pending_retrieval_steps) + 1}"
929
- self._pending_retrieval_steps.append(step)
933
+ with self._state_lock:
934
+ if top and top in self._runs:
935
+ # Retriever ran inside an active chain - attach directly
936
+ retrievals = self._runs[top]["retrieval_steps"]
937
+ step["name"] = f"Retrieval {len(retrievals) + 1}"
938
+ retrievals.append(step)
939
+ else:
940
+ # Retriever ran before the chain started (pre-run RAG pattern)
941
+ step["name"] = f"Retrieval {len(self._pending_retrieval_steps) + 1}"
942
+ self._pending_retrieval_steps.append(step)
930
943
 
931
944
  # ------------------------------------------------------------------
932
945
  # Helpers
@@ -107,18 +107,51 @@ class AgentXLlamaIndexHandler(BaseCallbackHandler):
107
107
  name: str = "llamaindex-agent",
108
108
  metadata: Optional[Dict[str, Any]] = None,
109
109
  session_id: Optional[str] = None,
110
+ max_run_age_seconds: float = 900.0,
110
111
  ) -> None:
111
112
  super().__init__(event_starts_to_ignore=[], event_ends_to_ignore=[])
112
113
  self._tracer = tracer
113
114
  self._name = name
114
115
  self._metadata = metadata
115
116
  self._session_id = session_id
117
+ # Safety net mirroring langchain.py's _prune_stale_entries: state is
118
+ # normally popped in on_event_end, but an event whose end callback never
119
+ # fires (hard crash, integration bug) would leak forever in this
120
+ # long-lived singleton handler. Entries older than this are swept out
121
+ # at the top of on_event_start.
122
+ self._max_run_age_seconds = max_run_age_seconds
116
123
 
117
124
  self._parents: Dict[str, Optional[str]] = {}
118
125
  self._roots: Dict[str, bool] = {}
119
126
  self._runs: Dict[str, Dict[str, Any]] = {}
120
127
  self._starts: Dict[str, Dict[str, Any]] = {}
121
128
 
129
+ def _prune_stale_entries(self) -> None:
130
+ """Sweep out event_id entries older than max_run_age_seconds - see __init__'s comment."""
131
+ cutoff = time.time() - self._max_run_age_seconds
132
+
133
+ # Every live event_id has a _starts entry (set in on_event_start and
134
+ # popped with _parents/_roots in on_event_end), each carrying its own
135
+ # start timestamp.
136
+ stale_event_ids = [
137
+ event_id for event_id, info in self._starts.items() if info.get("start", 0) < cutoff
138
+ ]
139
+ for event_id in stale_event_ids:
140
+ self._starts.pop(event_id, None)
141
+ self._parents.pop(event_id, None)
142
+ self._roots.pop(event_id, None)
143
+ self._runs.pop(event_id, None)
144
+
145
+ # Root runs outlive their own _starts entry until the root's end event
146
+ # fires - sweep those by the run state's own start timestamp.
147
+ stale_run_ids = [
148
+ event_id for event_id, state in self._runs.items() if state.get("start", 0) < cutoff
149
+ ]
150
+ for event_id in stale_run_ids:
151
+ self._runs.pop(event_id, None)
152
+ self._roots.pop(event_id, None)
153
+ self._parents.pop(event_id, None)
154
+
122
155
  # ------------------------------------------------------------------
123
156
  # BaseCallbackHandler protocol
124
157
  # ------------------------------------------------------------------
@@ -138,6 +171,7 @@ class AgentXLlamaIndexHandler(BaseCallbackHandler):
138
171
  **kwargs: Any,
139
172
  ) -> str:
140
173
  payload = payload or {}
174
+ self._prune_stale_entries()
141
175
  self._parents[event_id] = parent_id
142
176
 
143
177
  root_id = self._find_root(parent_id)
@@ -251,11 +285,16 @@ class AgentXLlamaIndexHandler(BaseCallbackHandler):
251
285
  tool_output = payload.get(EventPayload.FUNCTION_OUTPUT)
252
286
  state["tool_call_steps"].append({
253
287
  "name": tool_name,
254
- "duration_ms": (end_t - start_t) * 1000,
288
+ # tracer._merge_child_run's tool_calls loop reads "latency_ms"
289
+ # (not "duration_ms" like execution/retrieval steps).
290
+ "latency_ms": int((end_t - start_t) * 1000),
255
291
  "start_time": start_t,
256
292
  "end_time": end_t,
257
293
  "input": _safe_serialize(tool_input) if tool_input is not None else None,
258
294
  "output": f"ERROR: {exception}" if exception else (str(tool_output) if tool_output is not None else None),
295
+ # The engine's failure test is success === false; without this a
296
+ # failed tool call would read as passing.
297
+ "success": exception is None,
259
298
  })
260
299
 
261
300
  if is_root:
@@ -277,10 +316,9 @@ class AgentXLlamaIndexHandler(BaseCallbackHandler):
277
316
  return None
278
317
 
279
318
  def _send_trace(self, state: Dict[str, Any]) -> None:
280
- # tool_call_steps entries carry start_time/end_time (unlike langchain.py's leaner
281
- # wire-shaped tool_calls list) - _merge_child_run's tool_calls loop falls back to
282
- # computing duration from those when no explicit latency_ms is present, so each tool
283
- # call still positions correctly in the tree panel instead of defaulting to offset 0.
319
+ # tool_call_steps entries carry latency_ms AND start_time/end_time - _merge_child_run's
320
+ # tool_calls loop reads latency_ms for duration and the timestamps for position, so each
321
+ # tool call lands correctly in the tree panel instead of defaulting to offset 0.
284
322
  with self._tracer.trace(
285
323
  self._name, metadata=self._metadata, session_id=self._session_id, framework="llamaindex"
286
324
  ) as span:
@@ -16,6 +16,8 @@ Requires: ``pip install "agentx-python[openai-agents]"``
16
16
  from __future__ import annotations
17
17
 
18
18
  from datetime import datetime, timezone
19
+ import time
20
+ from uuid import uuid4
19
21
  from typing import Any, Dict, List, Optional
20
22
 
21
23
  from agentx.tracing.tracer import Tracer, _safe_serialize
@@ -145,7 +147,15 @@ class AgentXTracingProcessor:
145
147
  metadata=self._metadata,
146
148
  session_id=self._session_id,
147
149
  )
148
- root_span.__enter__()
150
+ # Deliberately NOT root_span.__enter__(): enter pushes onto the CALLING thread's
151
+ # active-span stack, but the Agents SDK fires on_trace_end on whatever thread it
152
+ # likes - the pop then no-ops there, the entry never drains, and every later
153
+ # unrelated trace on this thread is mis-filed as a child of this dead run (and
154
+ # inherits its session). Start time and session are set by hand instead; on_span_end
155
+ # already parents via child_span() on this exact reference, no stack involved.
156
+ root_span._start = time.time()
157
+ if root_span._session_id is None:
158
+ root_span._session_id = f"sdk_{uuid4().hex}"
149
159
  self._spans[trace_id] = {
150
160
  "root_span": root_span,
151
161
  "llm_call_count": 0,
@@ -173,6 +183,9 @@ class AgentXTracingProcessor:
173
183
  root_span._output_tokens = state["output_tokens"]
174
184
  if state.get("error"):
175
185
  root_span.set_error(state["error"])
186
+ # Close WITHOUT touching the thread-local stack (see on_trace_start). __exit__'s only
187
+ # stack interaction is the pop, which is a no-op for a never-pushed span - safe to call
188
+ # directly for its send/flush behavior.
176
189
  root_span.__exit__(None, None, None)
177
190
 
178
191
  def on_span_start(self, span: Any) -> None:
agentx/monitor/client.py CHANGED
@@ -28,7 +28,12 @@ _RETRY_BACKOFF = [1.0, 2.0, 4.0]
28
28
 
29
29
 
30
30
  class AgentXMonitorError(Exception):
31
- pass
31
+ """``status_code`` carries the HTTP status when the error came from a server
32
+ response; it is ``None`` for transport-level failures and retry exhaustion."""
33
+
34
+ def __init__(self, message: str, status_code: Optional[int] = None) -> None:
35
+ super().__init__(message)
36
+ self.status_code = status_code
32
37
 
33
38
 
34
39
  class AgentXAuthError(AgentXMonitorError):
@@ -208,17 +213,17 @@ class MonitorClient:
208
213
  continue
209
214
 
210
215
  if resp.status_code == 401:
211
- raise AgentXAuthError("Invalid or missing API key")
216
+ raise AgentXAuthError("Invalid or missing API key", status_code=401)
212
217
  if resp.status_code == 422:
213
- raise AgentXValidationError(resp.text)
218
+ raise AgentXValidationError(resp.text, status_code=422)
214
219
  if retry and resp.status_code in _RETRYABLE_STATUS and attempt < _MAX_RETRIES - 1:
215
220
  logger.debug(
216
221
  "Retryable status %d (attempt %d)", resp.status_code, attempt + 1
217
222
  )
218
- last_exc = AgentXMonitorError(f"HTTP {resp.status_code}")
223
+ last_exc = AgentXMonitorError(f"HTTP {resp.status_code}", status_code=resp.status_code)
219
224
  continue
220
225
  if not resp.ok:
221
- raise AgentXMonitorError(f"HTTP {resp.status_code}: {resp.text}")
226
+ raise AgentXMonitorError(f"HTTP {resp.status_code}: {resp.text}", status_code=resp.status_code)
222
227
  try:
223
228
  return resp.json()
224
229
  except Exception:
@@ -481,7 +486,7 @@ class MonitorClient:
481
486
  def propose_online_evaluator_tuning(self, evaluator_id: str, window: str = "7d") -> dict:
482
487
  data = self._request(
483
488
  "POST", f"/agent-monitoring/online-evaluators/{evaluator_id}/tune",
484
- base=self._api_root(), json={"window": window}, timeout=300,
489
+ base=self._api_root(), json={"window": window}, timeout=300, retry=False,
485
490
  )
486
491
  return data.get("proposal", data) if isinstance(data, dict) else data
487
492
 
@@ -490,7 +495,7 @@ class MonitorClient:
490
495
  ) -> dict:
491
496
  return self._request(
492
497
  "POST", f"/agent-monitoring/online-evaluators/{evaluator_id}/tune/validate",
493
- base=self._api_root(), json={**criteria, "window": window}, timeout=600,
498
+ base=self._api_root(), json={**criteria, "window": window}, timeout=600, retry=False,
494
499
  )
495
500
 
496
501
  def publish_online_evaluator_tuning(
@@ -505,7 +510,7 @@ class MonitorClient:
505
510
  payload["force"] = True
506
511
  return self._request(
507
512
  "POST", f"/agent-monitoring/online-evaluators/{evaluator_id}/tune/publish",
508
- base=self._api_root(), json=payload, timeout=60,
513
+ base=self._api_root(), json=payload, timeout=60, retry=False,
509
514
  )
510
515
 
511
516
  def update_profile(self, agent_id: str, payload: dict) -> MonitorProfile:
@@ -49,7 +49,8 @@ class ScorerGroupsClient:
49
49
  )
50
50
  if response.status_code >= 400:
51
51
  raise AgentXScorerGroupsError(f"HTTP {response.status_code}: {response.text}")
52
- return response.json()
52
+ # DELETE (and any other empty 2xx) has no body - .json() on it raises.
53
+ return response.json() if response.text else {}
53
54
 
54
55
  def list(self) -> List[ScorerGroup]:
55
56
  return [ScorerGroup(g) for g in self._request("GET", self._base).get("scorerGroups", [])]
agentx/monitor/scorers.py CHANGED
@@ -65,11 +65,18 @@ class ScorersClient:
65
65
  return [p for p in patterns if p.get("source") == "builtIn"]
66
66
 
67
67
  def _enabled_template_keys(self) -> List[str]:
68
- return [p["key"] for p in self.templates() if p.get("enabled")]
68
+ # .get("key"): defensive against a template row missing its key (the wire owns this
69
+ # shape, not the SDK) - a keyless row is skipped rather than KeyError-ing the sweep.
70
+ return [p.get("key") for p in self.templates() if p.get("enabled") and p.get("key")]
69
71
 
70
72
  def enable(self, keys: Sequence[str]) -> List[str]:
71
73
  """Enable template scorers by key (e.g. ``["pii-in-response"]``), preserving what is
72
- already on. Returns the resulting enabled-key list."""
74
+ already on. Returns the resulting enabled-key list.
75
+
76
+ Note: enable()/disable() are a read-modify-write over the project's single
77
+ enabledBuiltinPatterns list - two concurrent callers (or a dashboard edit racing an
78
+ SDK call) can lose one side's change. There is no engine-side merge; serialize
79
+ catalog edits if that matters."""
73
80
  merged = sorted(set(self._enabled_template_keys()) | set(keys))
74
81
  self._request("PUT", "/settings/monitoring-defaults", json={"enabledBuiltinPatterns": merged})
75
82
  return merged
@@ -27,8 +27,24 @@ class MonitorSessionClient:
27
27
  Baseline Judge existed - branch defensively on unknown kinds."""
28
28
  return self._client.list_session_scores(session_id)
29
29
 
30
+ def judge(self, session_id: str, evaluator_id: str, *, if_stale: bool = False) -> dict:
31
+ """Judge one session with one session-scoped evaluator, now (one judge call).
32
+ ``if_stale=True`` skips re-judging a session that was already scored since its
33
+ last activity - the engine then answers ``{"skipped": True}`` without spending
34
+ another judge call. Returns the score row (or that skip marker)."""
35
+ data = self._client._request(
36
+ "POST",
37
+ f"/agent-monitoring/sessions/{session_id}/judge/{evaluator_id}",
38
+ base=self._client._api_root(),
39
+ timeout=120,
40
+ retry=False,
41
+ params={"ifStale": "true"} if if_stale else None,
42
+ )
43
+ return data.get("score", data) if isinstance(data, dict) else data
44
+
30
45
  def run_sweep(self) -> dict:
31
46
  """Trigger the idle-session sweep once (normally automatic, every minute) - scores
32
47
  idle multi-turn sessions with every enabled session-scoped evaluator and scorer
33
- group. Returns ``{"judged": n}``."""
48
+ group. Returns ``{"judged": n}``; the response may instead carry ``skipped: true``
49
+ when another sweep is already in flight."""
34
50
  return self._client.run_session_sweep()
@@ -1,14 +1,20 @@
1
- """Dataclasses for CI/CD evaluation run responses."""
1
+ """Dataclasses for CI/CD evaluation run responses.
2
+
3
+ Annotations deliberately use typing.Optional/List (not PEP 604/585 syntax):
4
+ ``from __future__ import annotations`` only defers evaluation, and
5
+ ``typing.get_type_hints`` on these classes still has to resolve the strings
6
+ at runtime, which fails for ``str | None`` / ``list[...]`` on Python 3.9.
7
+ """
2
8
  from __future__ import annotations
3
9
 
4
10
  from dataclasses import dataclass, field
5
- from typing import Any, Literal
11
+ from typing import Any, Dict, List, Literal, Optional
6
12
 
7
13
 
8
14
  @dataclass
9
15
  class CITestCase:
10
16
  index: int
11
- query: str | None = None # None when ci.exposeTestInputs is false
17
+ query: Optional[str] = None # None when ci.exposeTestInputs is false
12
18
 
13
19
 
14
20
  @dataclass
@@ -16,7 +22,7 @@ class CIRun:
16
22
  run_id: str
17
23
  dataset_id: str
18
24
  total_questions: int
19
- test_cases: list[CITestCase]
25
+ test_cases: List[CITestCase]
20
26
  expires_at: str
21
27
 
22
28
 
@@ -47,20 +53,20 @@ class CIRunResult:
47
53
  pass_rate: float
48
54
  total_questions: int
49
55
  passed_questions: int
50
- scores: list[CIQuestionScore] = field(default_factory=list)
51
- violations: list[ThresholdViolation] = field(default_factory=list)
52
- git_context: dict | None = None
53
- finalized_at: str | None = None
56
+ scores: List[CIQuestionScore] = field(default_factory=list)
57
+ violations: List[ThresholdViolation] = field(default_factory=list)
58
+ git_context: Optional[Dict[str, Any]] = None
59
+ finalized_at: Optional[str] = None
54
60
 
55
61
 
56
62
  @dataclass
57
63
  class CIRunStatus:
58
64
  run_id: str
59
65
  status: Literal["in_progress", "completed", "failed"]
60
- gate: Literal["pass", "fail"] | None
66
+ gate: Optional[Literal["pass", "fail"]]
61
67
  results_submitted: int
62
68
  total_questions: int
63
69
  created_at: str
64
70
  expires_at: str
65
- finalized_at: str | None = None
66
- git_context: dict | None = None
71
+ finalized_at: Optional[str] = None
72
+ git_context: Optional[Dict[str, Any]] = None
@@ -138,27 +138,42 @@ class IngestClient:
138
138
  """
139
139
  Send a trace payload synchronously and return the ingested trace's id, or ``None`` on
140
140
  failure. Used by ``Tracer.trace(..., sync=True)`` when the caller needs the trace_id back
141
- immediately (e.g. to attach it to an evaluation result) - unlike ``enqueue()``, this blocks
142
- and does not retry, trading the tracer's usual fire-and-forget guarantee for a same-call
143
- result. Never raises; a failed send just means no trace_id (never blocks the caller's eval
144
- run over a tracing hiccup).
141
+ immediately (e.g. to attach it to an evaluation result) - unlike ``enqueue()``, this
142
+ blocks, trading the tracer's usual fire-and-forget guarantee for a same-call result.
143
+ A 429/503 (engine shedding load or briefly unavailable) is retried up to 2 times,
144
+ honoring the server's Retry-After (capped at 5s per wait) - span ids make redelivery
145
+ idempotent server-side, so a retry can never double-ingest. Never raises; ``None``
146
+ means the trace was NOT stored (there is no local persistence or background retry
147
+ beyond those brief attempts), so no trace_id exists for it.
145
148
  """
146
149
  if self._workspace_id:
147
150
  payload = {**payload, "workspaceId": self._workspace_id}
148
- try:
149
- resp = self._session.post(self._endpoint, json=payload, timeout=10)
150
- except requests.RequestException as exc:
151
- self._warn_delivery(f"{exc.__class__.__name__}: {exc}")
152
- logger.debug("agentx ingest sync send error: %s", exc)
153
- return None
154
- if not resp.ok:
155
- self._warn_delivery(f"HTTP {resp.status_code}", status=resp.status_code)
156
- logger.debug("agentx ingest sync HTTP %d: %s", resp.status_code, resp.text[:200])
157
- return None
158
- try:
159
- return resp.json().get("trace_id")
160
- except Exception:
161
- return None
151
+ for attempt in range(3): # 1 try + up to 2 bounded retries on 429/503
152
+ try:
153
+ resp = self._session.post(self._endpoint, json=payload, timeout=10)
154
+ except requests.RequestException as exc:
155
+ self._warn_delivery(f"{exc.__class__.__name__}: {exc}")
156
+ logger.debug("agentx ingest sync send error: %s", exc)
157
+ return None
158
+ if resp.status_code in (429, 503) and attempt < 2:
159
+ retry_after = resp.headers.get("Retry-After")
160
+ wait = 1.0
161
+ if retry_after:
162
+ try:
163
+ wait = min(5.0, float(retry_after))
164
+ except ValueError:
165
+ pass
166
+ time.sleep(wait)
167
+ continue
168
+ if not resp.ok:
169
+ self._warn_delivery(f"HTTP {resp.status_code}", status=resp.status_code)
170
+ logger.debug("agentx ingest sync HTTP %d: %s", resp.status_code, resp.text[:200])
171
+ return None
172
+ try:
173
+ return resp.json().get("trace_id")
174
+ except Exception:
175
+ return None
176
+ return None # pragma: no cover - loop always returns
162
177
 
163
178
  def send_trace_sync_detailed(self, payload: Dict[str, Any]) -> Optional[Dict[str, Any]]:
164
179
  """``send_trace_sync`` returning the full response body instead of just the id - the
@@ -418,16 +433,22 @@ class IngestClient:
418
433
 
419
434
  def _send(self, payload: Dict[str, Any]) -> None:
420
435
  last_exc: Optional[Exception] = None
421
- for attempt, wait in enumerate([0.0] + _RETRY_BACKOFF):
422
- if wait:
436
+ schedule = [0.0] + _RETRY_BACKOFF
437
+ skip_next_wait = False
438
+ for attempt, wait in enumerate(schedule):
439
+ if wait and not skip_next_wait:
423
440
  time.sleep(wait)
441
+ skip_next_wait = False
424
442
  try:
425
443
  resp = self._session.post(self._endpoint, json=payload, timeout=10)
426
444
  except requests.RequestException as exc:
427
445
  last_exc = exc
428
446
  continue
429
447
 
430
- if resp.status_code in _RETRYABLE_STATUS and attempt < _MAX_RETRIES - 1:
448
+ # Gate on the schedule itself so HTTP-status retries walk the SAME full backoff
449
+ # schedule connection errors do (the old `attempt < _MAX_RETRIES - 1` gate left the
450
+ # schedule's last backoff entry unreachable for HTTP retries).
451
+ if resp.status_code in _RETRYABLE_STATUS and attempt < len(schedule) - 1:
431
452
  # 429 = the engine's bounded ingest queue shedding load (its ADR-0005): honor
432
453
  # Retry-After exactly instead of the generic backoff schedule, so the SDK backs
433
454
  # off in step with the server's own flush cadence.
@@ -435,6 +456,9 @@ class IngestClient:
435
456
  if resp.status_code == 429 and retry_after:
436
457
  try:
437
458
  time.sleep(min(30.0, float(retry_after)))
459
+ # Retry-After REPLACES the schedule's next wait - sleeping both would
460
+ # back off longer than either the server or the schedule asked for.
461
+ skip_next_wait = True
438
462
  except ValueError:
439
463
  pass
440
464
  last_exc = Exception(f"HTTP {resp.status_code}")
@@ -442,6 +466,9 @@ class IngestClient:
442
466
  if not resp.ok:
443
467
  self._warn_delivery(f"HTTP {resp.status_code}", status=resp.status_code)
444
468
  logger.debug("agentx ingest HTTP %d: %s", resp.status_code, resp.text[:200])
469
+ # Non-retryable failure still cost us this payload - never drop silently
470
+ # (same rule _drain and the retries-exhausted path below follow).
471
+ self._record_drop(f"HTTP {resp.status_code}")
445
472
  return
446
473
  return
447
474
 
agentx/tracing/tracer.py CHANGED
@@ -4,6 +4,7 @@ import asyncio
4
4
  import concurrent.futures
5
5
  import functools
6
6
  import inspect
7
+ import contextvars
7
8
  import threading
8
9
  import time
9
10
  from contextlib import contextmanager
@@ -70,7 +71,7 @@ class _TraceSpan:
70
71
  model: Optional[str] = None,
71
72
  session_id: Optional[str] = None,
72
73
  sync: bool = False,
73
- monitor: bool = False,
74
+ monitor: Optional[bool] = None,
74
75
  pattern_ids: Optional[List[str]] = None,
75
76
  agent_id: Optional[str] = None,
76
77
  span_kind: Optional[str] = None,
@@ -104,6 +105,11 @@ class _TraceSpan:
104
105
  # __enter__/_merge_child_run/child_span.
105
106
  self._span_id = uuid4().hex
106
107
  self._parent_span_id: Optional[str] = None
108
+ # Set on first __enter__ - a re-entered span object regenerates its
109
+ # span_id there so each `with span:` re-use sends a fresh identity
110
+ # (the server dedupes on span_id, so a reused id would silently
111
+ # collapse the second run into the first).
112
+ self._entered = False
107
113
  # Numbers auto-named "LLM Call N"/"Retrieval N" child spans - see _merge_child_run.
108
114
  self._child_span_count = 0
109
115
  # Monitor: True checks this trace against patterns immediately on ingest, no dashboard
@@ -145,6 +151,13 @@ class _TraceSpan:
145
151
  # ------------------------------------------------------------------
146
152
 
147
153
  def __enter__(self) -> "_TraceSpan":
154
+ if self._entered:
155
+ # Re-using one span object for another `with` block: regenerate the
156
+ # identity so this run sends its own span row instead of being
157
+ # deduped server-side against the first entry's span_id.
158
+ self._span_id = uuid4().hex
159
+ self._trace_id = None
160
+ self._entered = True
148
161
  self._start = time.time()
149
162
  # Resolve real span hierarchy against whatever's currently active on this thread, before
150
163
  # pushing self (so `parent` here is the actual enclosing span, not self).
@@ -581,6 +594,14 @@ class _RetrievalRecorder:
581
594
  self.output: Any = None
582
595
 
583
596
 
597
+ class _MemoryOpRecorder:
598
+ """Handle yielded by ``Tracer.trace_memory()`` - set ``output`` (what was recalled or
599
+ stored) inside the block."""
600
+
601
+ def __init__(self) -> None:
602
+ self.output: Any = None
603
+
604
+
584
605
  class _ToolCallRecorder:
585
606
  """Handle yielded by ``Tracer.trace_tool_call()`` - set ``output`` inside the block.
586
607
  ``success``/``error`` may be set manually; an exception escaping the block sets them
@@ -619,36 +640,39 @@ class Tracer:
619
640
  self._client = ingest_client
620
641
  self._pending_tool_calls: List[Dict[str, Any]] = []
621
642
  self._pending_retrievals: List[Dict[str, Any]] = []
622
- self._local = threading.local()
643
+ # Context-local, not thread-local: two coroutines interleaving on one event loop each
644
+ # get their own asyncio task Context, so concurrent `async def` agents no longer
645
+ # mis-parent each other's spans (a thread-local stack merged them into one fabricated
646
+ # tree). Bare threads keep the old behavior - each starts an empty Context. Stored
647
+ # immutably (tuple, copy-on-write) so a child task's pushes never leak into siblings.
648
+ self._span_stack_var: "contextvars.ContextVar[tuple]" = contextvars.ContextVar(
649
+ f"agentx_span_stack_{id(self)}", default=()
650
+ )
623
651
 
624
652
  # ------------------------------------------------------------------
625
- # Active-span stack (per thread) - lets auto-instrumented integrations
653
+ # Active-span stack (per context) - lets auto-instrumented integrations
626
654
  # (e.g. patch_anthropic_client) detect they're running inside a
627
655
  # `with tracer.trace(...)` block and attach to it as an LLM-call step
628
656
  # instead of sending their own independent trace.
629
657
  # ------------------------------------------------------------------
630
658
 
631
- def _get_span_stack(self) -> List["_TraceSpan"]:
632
- stack = getattr(self._local, "span_stack", None)
633
- if stack is None:
634
- stack = []
635
- self._local.span_stack = stack
636
- return stack
659
+ def _get_span_stack(self) -> tuple:
660
+ return self._span_stack_var.get()
637
661
 
638
662
  def _push_active_span(self, span: "_TraceSpan") -> None:
639
- self._get_span_stack().append(span)
663
+ self._span_stack_var.set(self._span_stack_var.get() + (span,))
640
664
 
641
665
  def _pop_active_span(self, span: "_TraceSpan") -> None:
642
- stack = self._get_span_stack()
666
+ stack = self._span_stack_var.get()
643
667
  if stack and stack[-1] is span:
644
- stack.pop()
668
+ self._span_stack_var.set(stack[:-1])
645
669
  elif span in stack:
646
- stack.remove(span)
670
+ self._span_stack_var.set(tuple(item for item in stack if item is not span))
647
671
 
648
672
  @property
649
673
  def current_span(self) -> Optional["_TraceSpan"]:
650
- """The innermost ``with tracer.trace(...)`` span active on this thread, if any."""
651
- stack = self._get_span_stack()
674
+ """The innermost ``with tracer.trace(...)`` span active in this context, if any."""
675
+ stack = self._span_stack_var.get()
652
676
  return stack[-1] if stack else None
653
677
 
654
678
  @contextmanager
@@ -832,6 +856,92 @@ class Tracer:
832
856
  span_kind="retrieval",
833
857
  )
834
858
 
859
+ def record_memory(
860
+ self,
861
+ name: str = "Memory",
862
+ *,
863
+ operation: Optional[str] = None,
864
+ query: Optional[str] = None,
865
+ output: Any = None,
866
+ duration_ms: Optional[float] = None,
867
+ start_time: Optional[float] = None,
868
+ end_time: Optional[float] = None,
869
+ ) -> None:
870
+ """
871
+ Manually record a long-term-memory operation (a Mem0/Zep/Letta-style recall or store)
872
+ as a ``span_kind="memory"`` child span of the active span. ``operation`` is free text -
873
+ conventionally ``"read"`` or ``"write"`` - carried in the span's metadata; the kind
874
+ itself stays one value so dashboards and scorers can select all memory activity at once.
875
+ Deliberately NOT a retrieval: retrieval spans feed the RAG judges' ``{context}``
876
+ (knowledge grounding), while memory is recalled state - see the engine's spanKind.ts.
877
+ """
878
+ active_span = self.current_span
879
+ if active_span is None:
880
+ # Same posture as record_retrieval: queue and merge into the next trace this
881
+ # tracer sends (the patched-client flow where the memory op runs just before a
882
+ # standalone completions call) instead of silently dropping the record.
883
+ latency_ms = (
884
+ int(duration_ms)
885
+ if duration_ms is not None
886
+ else int((end_time - start_time) * 1000)
887
+ if start_time is not None and end_time is not None
888
+ else None
889
+ )
890
+ self._pending_retrievals.append({
891
+ "name": name,
892
+ "query": _safe_serialize(query) if query is not None else None,
893
+ "output": _safe_serialize(output) if output is not None else None,
894
+ "duration_ms": latency_ms,
895
+ "kind": "memory",
896
+ **({"operation": operation} if operation else {}),
897
+ })
898
+ return
899
+ active_span.child_span(
900
+ name,
901
+ start_time=start_time,
902
+ end_time=end_time,
903
+ duration_ms=duration_ms,
904
+ input=query,
905
+ output=output,
906
+ metadata={"kind": "memory", **({"operation": operation} if operation else {})},
907
+ span_kind="memory",
908
+ )
909
+
910
+ @contextmanager
911
+ def trace_memory(
912
+ self, name: str = "Memory", *, operation: Optional[str] = None, query: Optional[str] = None
913
+ ) -> Iterator["_MemoryOpRecorder"]:
914
+ """
915
+ Context manager that times a memory operation and records it via
916
+ :meth:`record_memory` on exit::
917
+
918
+ with tracer.trace_memory("user prefs", operation="read", query=user_id) as m:
919
+ m.output = memory.search(user_id, question)
920
+ """
921
+ start_t = time.time()
922
+ recorder = _MemoryOpRecorder()
923
+ error: Optional[str] = None
924
+ try:
925
+ yield recorder
926
+ except Exception as exc:
927
+ # A memory op that raised must not be recorded as a clean span (trace_tool_call
928
+ # precedent) - fold the error into the output and re-raise.
929
+ error = str(exc)
930
+ raise
931
+ finally:
932
+ end_t = time.time()
933
+ if error is not None and recorder.output is None:
934
+ recorder.output = f"ERROR: {error}"
935
+ self.record_memory(
936
+ name,
937
+ operation=operation,
938
+ query=query,
939
+ output=recorder.output,
940
+ duration_ms=(end_t - start_t) * 1000,
941
+ start_time=start_t,
942
+ end_time=end_t,
943
+ )
944
+
835
945
  @contextmanager
836
946
  def trace_retrieval(self, name: str = "Retrieval", *, query: Optional[str] = None) -> Iterator["_RetrievalRecorder"]:
837
947
  """
agentx/version.py CHANGED
@@ -1,7 +1,7 @@
1
- VERSION = "0.8.18"
1
+ VERSION = "0.8.20"
2
2
 
3
3
  # The AgentX-trace-eval release this SDK version is tested against - what `agentx-trace-eval`
4
4
  # installs and converges to (see agentx/cli.py). Bump together with VERSION when releasing, so
5
5
  # every published SDK names a known-good engine+dashboard pair. Users can override with
6
6
  # AGENTX_TRACE_EVAL_VERSION=<tag|latest>.
7
- ENGINE_VERSION = "v0.3.18"
7
+ ENGINE_VERSION = "v0.3.21"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentx-python
3
- Version: 0.8.18
3
+ Version: 0.8.20
4
4
  Summary: Official Python SDK for AgentX (https://www.agentx.so/)
5
5
  Home-page: https://github.com/AgentX-ai/AgentX-python
6
6
  Author: Robin Wang and AgentX Team
@@ -1,4 +1,4 @@
1
- agentx/__init__.py,sha256=4iMiGLNU4S-I-WyX0Z7l0BFUY_krVJ3seMQPqyO-ReA,602
1
+ agentx/__init__.py,sha256=k-89JLBl_jNXhVCtBbdtlDUGA1nLcDblvd7adk7Y3S8,790
2
2
  agentx/agentx.py,sha256=WuzVQtXSPqBhRwKxkUlQ8VG2cwvGpbMcti4n6aTNPwA,8816
3
3
  agentx/cli.py,sha256=yawLQLSeZ7KIl7ukgmly-lrYbx9GgKcaI9M4xamiMPw,10468
4
4
  agentx/exceptions.py,sha256=2tDLdZjriFBQbjgzyIcah12RazpbujhqZSRbSXgAml4,1377
@@ -10,17 +10,17 @@ agentx/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
10
10
  agentx/testing.py,sha256=0shZEid_vhJgBegpO_loUq6APYDxQrUgD3jvcRdIDV8,7372
11
11
  agentx/traces.py,sha256=sz9gxutlDKNgf0fdCkyzKsGwK5eMAJuIEl3NIwGrsuo,2287
12
12
  agentx/util.py,sha256=lt2Kpg4Fj8whrSuAhLEahcPEIYRVpK9RCC5y9ShdTJs,703
13
- agentx/version.py,sha256=12ZEc4OJ-zU1rtsHoDBC5wqnwX7BmZ9OmiL3F4NWnMQ,366
13
+ agentx/version.py,sha256=vPqx2---hI57dDK8pRv2qIEkiaUleIQleNKmcs_kTw4,366
14
14
  agentx/evaluations/__init__.py,sha256=Erv7RGFlRxGTG4rVb2uHhCqLLEX_iAWKqkZKzK6CumE,262
15
15
  agentx/evaluations/_term.py,sha256=WFpiNzdgDBeJJ-Gg-6X7TwwxllobuE4OqFUTuDQvS3Y,2529
16
- agentx/evaluations/client.py,sha256=f5GqsiuqknE-ENUYocBcXgV6OGMKdzmlsm76takJbwU,33112
17
- agentx/evaluations/datasets.py,sha256=DuX-wu5Kz953s6-605lt8N8-kn3F4WuwaCMq5ISZwbA,16051
16
+ agentx/evaluations/client.py,sha256=JPH4CAkPWz2S-8bnxoqAcB0T6Ml70tg6JUxf_ZyTHLM,34269
17
+ agentx/evaluations/datasets.py,sha256=YhwGMZkDqxfR-0e81o9WrQAq_Bc1BYclizh2ka1kBi0,16759
18
18
  agentx/evaluations/evaluation_settings.py,sha256=PhZtflmEAw33EHuzu4C_Vpm8BBjnIPw60y7VEJ7mU8s,6274
19
19
  agentx/evaluations/models.py,sha256=v-t6_HkEQlGNc7ZGfoYXZ7FrFjepuyRODkly-ce6Qjw,32936
20
20
  agentx/evaluations/prompts.py,sha256=8xyvMpAl3mD5xuSZozuaohdnCBWe5Q02RldwiPLUO20,3493
21
- agentx/evaluations/reporting.py,sha256=GtNnL-1eNEQrSqd0yrHi5mGqC9_u0kUAN8Mbjdhw6yY,6389
21
+ agentx/evaluations/reporting.py,sha256=RPeVBdfynQhkvk4KiOh6I-DqWFLfSrEniW3dWWTiW5I,6424
22
22
  agentx/evaluations/results.py,sha256=w7TRUik4eXIqrznXvlF_L9hnObnHvgdRKvihfN1XeAU,4692
23
- agentx/evaluations/runner.py,sha256=2lxbYKTE2OVH_DFq5yVdAH6cvW5rf3rL91VlEepWLiQ,34714
23
+ agentx/evaluations/runner.py,sha256=_P0ZyvR6k620jgKzv8JXDTklb9zXDv-qGF4KUug4RyA,35114
24
24
  agentx/evaluations/tool_schemas.py,sha256=ZyrnSOnx9xlSAj4swfDey9ZLS2ly-g21DByTwRAGwNA,2323
25
25
  agentx/evaluations/tracing.py,sha256=MSJD9bzfInMi7E70mc8mWUc0fx5DLkOgUzRru2mNQoc,1825
26
26
  agentx/evaluations/adapters/__init__.py,sha256=fK8Hx75usbiY03XUSmnLnSrwBXip2CQs24YZcdEgj6w,287
@@ -29,21 +29,21 @@ agentx/evaluations/adapters/precomputed.py,sha256=vxnQELmpiU2NkQATfiVjG7v1ZmmkrE
29
29
  agentx/evaluations/adapters/raw.py,sha256=FkDq_mdf21-xt5EHnGyIGcS6VZxkEkFcmM9VTd-T5sA,1130
30
30
  agentx/integrations/__init__.py,sha256=W-EcQhHE01-SGPeKTB2AgfvUyRzNP4BqUMMtP2qQ4Q0,647
31
31
  agentx/integrations/_traced_call.py,sha256=O6f8HKRHDfFwoUIZDtvENVsv_OsHwzuyPMbzBakKVdo,6662
32
- agentx/integrations/anthropic.py,sha256=Fdt1bVBKYuhUa8S4nsZLNqg4REUzaZsh-7RXEWuJ-0g,10392
33
- agentx/integrations/autogen.py,sha256=cZryUl45Sdr9t1ba4tkz45XfkvjCud1hPp6pV0auKy8,8863
34
- agentx/integrations/crewai.py,sha256=Ofid_YvIFYFoN12_gA3PjKAhlJgPNUh9sNyIbBcRkZs,10712
32
+ agentx/integrations/anthropic.py,sha256=z6o9cC9rBzxxbF1UBiMr-kjEgDUO-BCuPOFoqGN3qF4,10451
33
+ agentx/integrations/autogen.py,sha256=nl9XuhdBIajRmiz_DeXneqR-FtbslT0d9i1cZYKeJhA,9067
34
+ agentx/integrations/crewai.py,sha256=EFQuRfj-5bwxNTefZDLF9cCUY8TomS2qg9T4SmjayQk,11586
35
35
  agentx/integrations/databricks.py,sha256=vKXnur-LWMTzctFuIKte2N826sydXVlpi1qFXo2dfio,18044
36
36
  agentx/integrations/google_adk.py,sha256=swLpAfkuL2m1VZUSf_3ImwsrcTSxVdS5EIKRu2Tf4R4,12476
37
37
  agentx/integrations/google_genai.py,sha256=2patIdgyjEEW70FLP7ng4hcnFY-p0a1jPi1Pof28KBo,13272
38
- agentx/integrations/langchain.py,sha256=z4esU-IjWuXS0Ft7PPGxnTBxBrrN7JmKEqW_1HpTdMI,38745
38
+ agentx/integrations/langchain.py,sha256=IFfA7QddFPjK_JnFlruxrD-sCm5fgrY3VH_L4e4f3o0,39503
39
39
  agentx/integrations/litellm.py,sha256=F9Zcb58WtlBFsg_Ig1LGktpe99x9xXnYNi-C0UH54fI,6476
40
- agentx/integrations/llamaindex.py,sha256=pPw0N6VavIR25rrl6ZiUvb79hIPygfJgIxwr6kAvFqA,12777
40
+ agentx/integrations/llamaindex.py,sha256=nG3EVauSAmHPpyzZkyc31HL3cnTKHQCK8ypK7ZpoQDE,14707
41
41
  agentx/integrations/moveworks.py,sha256=IyBswLE5LwMSp1VbIvG5izYV5BvXmv4atZelnrDn8SE,25670
42
42
  agentx/integrations/openai.py,sha256=1KLs-hJaeW2emHPwNkmC3zcGksL3zd87uWbcLGRW704,6644
43
- agentx/integrations/openai_agents.py,sha256=tCa3kPxqux5LnNNDQ6zA05OybQrjkc_LeE08YuNPJEs,12742
43
+ agentx/integrations/openai_agents.py,sha256=9PIXFP3SuBJXc8f8ph7TBpoWoarfL265xRrYf0dgagc,13653
44
44
  agentx/monitor/__init__.py,sha256=MLiASSvQnCy2jqVNRTgPWFhUSf8lGYRQeEgEVVhE5Tg,574
45
45
  agentx/monitor/agents.py,sha256=8Xk4jWmNTvdtjFixHiLmw4jj96uqIXWIT6KW8zXoWrc,1006
46
- agentx/monitor/client.py,sha256=lQdTUsXcxZwUOhdxZ5X_GdlIGoAaZa55mwn2u9DmSaw,24008
46
+ agentx/monitor/client.py,sha256=Q-VQbSO8AQZS5t0hzHxblp2I6xV3rOEEYLewlnr8vgQ,24451
47
47
  agentx/monitor/improvement_groups.py,sha256=-6-FcJnyiTLqeh9z8D9wyFD1QFqEsdDdDZdDJhC6QXw,3712
48
48
  agentx/monitor/judge_scorers.py,sha256=HKCPW_9QJeyGEGwWxr_clep9vn27bwJTKnjN5lYbuSk,15972
49
49
  agentx/monitor/models.py,sha256=yTC3WTdTziHMkdeAXlTro-4YiSbBUHMgf9kNZHGfPDA,8470
@@ -52,23 +52,23 @@ agentx/monitor/patterns.py,sha256=NFHmHJqbq_KC3hNjh9lgv5aB8rsyUovGdd9YGwsQKfE,44
52
52
  agentx/monitor/profile.py,sha256=uGk-4bXT2vmnEQiY5u6rWhAfzZchHBXL74VD3piVHvU,3102
53
53
  agentx/monitor/review_queue.py,sha256=Tr3GDMJLJzyFKEEeB8HPmZBsNzzYVJYElDlfBC4qd68,3766
54
54
  agentx/monitor/rules.py,sha256=oLQ5RPgMPUKJFTSZzcBEMLzDlICElUNJTyIqxZMAGmQ,3040
55
- agentx/monitor/scorer_groups.py,sha256=qff9eheOU7jMM8wVKU7gRrOlafPiQv16ieywBLAXFH0,3734
56
- agentx/monitor/scorers.py,sha256=huIZBszrfLKI8I7VV6NFpW6piRsj9HI03rXUwWs2tgs,7336
57
- agentx/monitor/sessions.py,sha256=EnhyMMe2muwLyuKn99mLw-Ff1tHTMsgM7H_SXob1ltY,1570
55
+ agentx/monitor/scorer_groups.py,sha256=nye3v4Cs8qFS39sByiy0UbaOjdGPtVcoa8XC4vIDhZI,3838
56
+ agentx/monitor/scorers.py,sha256=ZEJCoQNx5u78qf6T9MEAyG2yufJIzxIzEU8PnYxL0Co,7844
57
+ agentx/monitor/sessions.py,sha256=bJOp-yBgnKczq2J8jsPSrstlViOkdGGN7ixH3oOYb7Q,2472
58
58
  agentx/monitor/signals.py,sha256=Ld11lW2lbg8NHBUHdgoY6iwCvJDK-eg4ge4qQ9y7nGY,1432
59
59
  agentx/resources/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
60
60
  agentx/resources/agent.py,sha256=ZKpxDYzJbKgOX4-W86U__b4gko7XlbwB4q6L8Az9uo0,2011
61
61
  agentx/resources/conversation.py,sha256=94pOdM6Pmj67RhikZVBNGkT3V-WKK_ZAD7s27p5-N0Y,4463
62
62
  agentx/resources/workforce.py,sha256=lfGVkoV9lcOp-lZScjTvwRarJMkhzfMfLm4syJDujGc,4168
63
63
  agentx/tracing/__init__.py,sha256=l2mIRILE-wWrCD0fekgrIOQUEvtDObzdrkgpQUtXroU,408
64
- agentx/tracing/ci_types.py,sha256=zFVcZGvc1qbVDdOlEPQ1i5R6terqJHxCzFF4ZjouxxI,1423
64
+ agentx/tracing/ci_types.py,sha256=b-W2LowRhNBVUdaoMPgd4bnKQy9AxhfiikftRwGPAqU,1778
65
65
  agentx/tracing/eval_scope.py,sha256=ElMbPxpqpQBVuaUnHNlw9RIQu71oyOpd0yR55bDH8eI,2144
66
66
  agentx/tracing/framework_detect.py,sha256=uV4O7Th-4_2UkdooAyaWCOIA0jwLJbblFQ_ucFDjeJM,2638
67
- agentx/tracing/ingest_client.py,sha256=spnm6bV3NXGuj2w7HwdKiJdKDJWRUFmSP3jl3rcGqkQ,18202
68
- agentx/tracing/tracer.py,sha256=TxY6zG6hbM81jpfrqP4imaEvoyy_hnJVUSTnoAVHMG0,56240
69
- agentx_python-0.8.18.dist-info/licenses/LICENSE,sha256=gZVsM-nLsE8vlaY6NXXsVoo6IlCClxkToAWmhgT3y_s,10762
70
- agentx_python-0.8.18.dist-info/METADATA,sha256=P0yCj9IwWgJrN6c_YcDBWj5YZeJCFaBkz_CNv1mJ2p4,22117
71
- agentx_python-0.8.18.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
72
- agentx_python-0.8.18.dist-info/entry_points.txt,sha256=rQqF1JTY3T1yfviBU1r8PU-5mBdfOk6P4bZ1L4BskIM,172
73
- agentx_python-0.8.18.dist-info/top_level.txt,sha256=s-q-HB9Gb_QdrZNacSeQyF_c25gQooMy7DlxzgLOHPk,7
74
- agentx_python-0.8.18.dist-info/RECORD,,
67
+ agentx/tracing/ingest_client.py,sha256=9Gs_XEE-VJxcWdXjm6buo8wj5l7Pva1gX6AvtFOdzZs,19926
68
+ agentx/tracing/tracer.py,sha256=iHQIR6XomOzYm9FvqSvu-hxM6Lg6mDXBMR8eTL-4ank,61289
69
+ agentx_python-0.8.20.dist-info/licenses/LICENSE,sha256=gZVsM-nLsE8vlaY6NXXsVoo6IlCClxkToAWmhgT3y_s,10762
70
+ agentx_python-0.8.20.dist-info/METADATA,sha256=_fBpNGWvcR0utTxdXRKZaFlhorOSqFXjSSim2KjW_Ro,22117
71
+ agentx_python-0.8.20.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
72
+ agentx_python-0.8.20.dist-info/entry_points.txt,sha256=rQqF1JTY3T1yfviBU1r8PU-5mBdfOk6P4bZ1L4BskIM,172
73
+ agentx_python-0.8.20.dist-info/top_level.txt,sha256=s-q-HB9Gb_QdrZNacSeQyF_c25gQooMy7DlxzgLOHPk,7
74
+ agentx_python-0.8.20.dist-info/RECORD,,