agentx-python 0.8.2__py3-none-any.whl → 0.8.4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,7 +1,7 @@
1
1
  from __future__ import annotations
2
2
 
3
3
  from typing import Any, Dict, List, Literal, Optional, Union
4
- from pydantic import BaseModel, Field, model_validator
4
+ from pydantic import AliasChoices, BaseModel, Field, model_validator
5
5
 
6
6
  # ---------------------------------------------------------------------------
7
7
  # Observable trace
@@ -418,7 +418,11 @@ class RunResultRow(BaseModel):
418
418
  latency_ms: Optional[float] = Field(default=None, alias="latencyMs")
419
419
  input_tokens: Optional[int] = Field(default=None, alias="inputTokens")
420
420
  output_tokens: Optional[int] = Field(default=None, alias="outputTokens")
421
- cosine_similarity: Optional[float] = Field(default=None, alias="cosineSimilarity")
421
+ # Self-host sends vectorSimilarity, the hosted platform cosineSimilarity - accept both
422
+ # (row.cosine_similarity was silently None forever on self-host before this).
423
+ cosine_similarity: Optional[float] = Field(
424
+ default=None, validation_alias=AliasChoices("cosineSimilarity", "vectorSimilarity")
425
+ )
422
426
  jaccard_similarity: Optional[float] = Field(default=None, alias="jaccardSimilarity")
423
427
  bleu_score: Optional[float] = Field(default=None, alias="bleuScore")
424
428
  rouge_score: Optional[float] = Field(default=None, alias="rougeScore")
@@ -468,7 +472,11 @@ class ReportStatistics(BaseModel):
468
472
  average_rating: float = Field(default=0.0, alias="averageRating")
469
473
  min_rating: float = Field(default=0.0, alias="minRating")
470
474
  max_rating: float = Field(default=0.0, alias="maxRating")
471
- cosine_similarity: Optional[float] = Field(default=None, alias="cosineSimilarity")
475
+ # Self-host sends vectorSimilarity, the hosted platform cosineSimilarity - accept both
476
+ # (row.cosine_similarity was silently None forever on self-host before this).
477
+ cosine_similarity: Optional[float] = Field(
478
+ default=None, validation_alias=AliasChoices("cosineSimilarity", "vectorSimilarity")
479
+ )
472
480
  jaccard_similarity: Optional[float] = Field(default=None, alias="jaccardSimilarity")
473
481
  bleu_score: Optional[float] = Field(default=None, alias="bleuScore")
474
482
  rouge_score: Optional[float] = Field(default=None, alias="rougeScore")
@@ -119,7 +119,21 @@ class EvaluationRunContext:
119
119
  # ------------------------------------------------------------------
120
120
 
121
121
  def execute(self, adapter: AdapterLike) -> "EvaluationRunContext":
122
- """Run all cases locally and submit batches to AgentX."""
122
+ """Run all cases locally and submit batches to AgentX.
123
+
124
+ The whole loop runs inside the eval-run scope (tracing/eval_scope.py): any trace the
125
+ agent function creates is stamped source="eval-run" + monitor=False automatically, so
126
+ eval traffic never skews production monitoring and no one has to remember a flag.
127
+ """
128
+ from agentx.tracing.eval_scope import enter_eval_run, exit_eval_run
129
+
130
+ scope_token = enter_eval_run(self._run.run_id)
131
+ try:
132
+ return self._execute_inner(adapter)
133
+ finally:
134
+ exit_eval_run(scope_token)
135
+
136
+ def _execute_inner(self, adapter: AdapterLike) -> "EvaluationRunContext":
123
137
  normalized = _wrap_adapter(adapter)
124
138
  cases = _build_cases(self._dataset, self._run, self._evaluation_settings)
125
139
  max_batch = self._run.limits.max_batch_size
@@ -0,0 +1,44 @@
1
+ """The eval-run scope: how traces created inside an evaluation stop passing as production.
2
+
3
+ An offline run executes the user's own agent function, and an instrumented agent traces itself -
4
+ which is exactly what makes trajectory matching and retrieval-context extraction work. But those
5
+ traces are not production traffic, and before this scope existed the burden of saying so sat on
6
+ every caller: remember ``monitor=False`` on every ``tracer.trace(...)`` inside an eval, or the
7
+ engine would double-judge each case, raise signals on synthetic questions, and count the run's
8
+ latencies into production KPIs. Nobody remembered - including our own samples.
9
+
10
+ ``EvaluationRunContext.execute()`` enters this scope around the whole run. While it is active,
11
+ every trace the tracer sends is stamped:
12
+
13
+ - ``source="eval-run"`` - the engine files it as eval traffic (excluded from monitor KPIs,
14
+ metrics, sessions and the Live Traces default view; cost keeps it, split out)
15
+ - ``monitor=False`` - unless the caller explicitly passed ``monitor=True``, which is
16
+ respected as a deliberate choice
17
+ - ``metadata.evalRunId`` - so a trace can always be walked back to the run that produced it
18
+
19
+ A ``contextvars.ContextVar`` rather than tracer state: it nests correctly, cannot leak across
20
+ concurrent runs in async code, and costs nothing when no run is active. The one known limit is
21
+ threads the agent function spawns itself - a context var does not cross a bare ``Thread()`` -
22
+ which matches the tracer's existing documented posture for user-managed threads.
23
+ """
24
+
25
+ from contextvars import ContextVar
26
+ from typing import Optional
27
+
28
+ EVAL_RUN_SOURCE = "eval-run"
29
+
30
+ _current_eval_run_id: ContextVar[Optional[str]] = ContextVar("agentx_eval_run_id", default=None)
31
+
32
+
33
+ def enter_eval_run(run_id: str):
34
+ """Mark the current context as inside an eval run. Returns the token for ``exit_eval_run``."""
35
+ return _current_eval_run_id.set(run_id)
36
+
37
+
38
+ def exit_eval_run(token) -> None:
39
+ _current_eval_run_id.reset(token)
40
+
41
+
42
+ def current_eval_run_id() -> Optional[str]:
43
+ """The run id when inside ``execute()``, else None."""
44
+ return _current_eval_run_id.get()
agentx/tracing/tracer.py CHANGED
@@ -13,6 +13,7 @@ from uuid import uuid4
13
13
  from agentx.exceptions import CIGateFailure
14
14
  from agentx.tracing.ingest_client import IngestClient
15
15
  from agentx.tracing.ci_types import CIRun, CIRunResult, CIRunStatus, CIQuestionScore
16
+ from agentx.tracing.eval_scope import EVAL_RUN_SOURCE, current_eval_run_id
16
17
 
17
18
  F = TypeVar("F", bound=Callable[..., Any])
18
19
 
@@ -167,9 +168,21 @@ class _TraceSpan:
167
168
  # flush() uses; child-only spans keep their async fire-and-forget behavior.
168
169
  self._tracer.flush(timeout=5.0)
169
170
 
171
+ # Inside an eval run (evaluations' execute()), every trace states what it is: eval
172
+ # traffic. monitor=False unless the caller explicitly said True; the run id rides in
173
+ # metadata so the trace can be walked back to its run. See tracing/eval_scope.py.
174
+ eval_run_id = current_eval_run_id()
175
+ monitor = self._monitor
176
+ metadata = self._metadata
177
+ source = None
178
+ if eval_run_id is not None:
179
+ source = EVAL_RUN_SOURCE
180
+ if monitor is not True:
181
+ monitor = False
182
+ metadata = {**(metadata or {}), "evalRunId": eval_run_id}
170
183
  self._trace_id = self._tracer._send(
171
184
  sync=self._sync,
172
- monitor=self._monitor,
185
+ monitor=monitor,
173
186
  pattern_ids=self._pattern_ids,
174
187
  name=self.name,
175
188
  agent_id=self._agent_id,
@@ -177,7 +190,7 @@ class _TraceSpan:
177
190
  output=_safe_serialize(self.output) if self.output is not None else None,
178
191
  latency_ms=latency_ms,
179
192
  error=self._error,
180
- metadata=self._metadata,
193
+ metadata=metadata,
181
194
  framework=self._framework or self._captured_framework,
182
195
  model=self._model or self._captured_model,
183
196
  tool_calls=self.tool_calls or None,
@@ -189,6 +202,7 @@ class _TraceSpan:
189
202
  span_id=self._span_id,
190
203
  parent_span_id=self._parent_span_id,
191
204
  span_kind=self._span_kind,
205
+ source=source,
192
206
  started_at_unix_nano=str(int(self._start * 1_000_000_000)) if self._start else None,
193
207
  )
194
208
  return False # never suppress exceptions
@@ -332,6 +346,8 @@ class _TraceSpan:
332
346
  wire["span_kind"] = span_kind
333
347
  if child._session_id:
334
348
  wire["session_id"] = child._session_id
349
+ if current_eval_run_id() is not None:
350
+ wire["source"] = EVAL_RUN_SOURCE
335
351
  wire["span_id"] = child._span_id
336
352
  if child._parent_span_id:
337
353
  wire["parent_span_id"] = child._parent_span_id
@@ -1148,6 +1164,8 @@ class Tracer:
1148
1164
  wire["started_at_unix_nano"] = payload["started_at_unix_nano"]
1149
1165
  if "agent_id" in payload:
1150
1166
  wire["agent_id"] = payload["agent_id"]
1167
+ if "source" in payload:
1168
+ wire["source"] = payload["source"]
1151
1169
  if "span_kind" in payload:
1152
1170
  wire["span_kind"] = payload["span_kind"]
1153
1171
 
agentx/version.py CHANGED
@@ -1 +1 @@
1
- VERSION = "0.8.2"
1
+ VERSION = "0.8.4"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentx-python
3
- Version: 0.8.2
3
+ Version: 0.8.4
4
4
  Summary: Official Python SDK for AgentX (https://www.agentx.so/)
5
5
  Home-page: https://github.com/AgentX-ai/AgentX-python
6
6
  Author: Robin Wang and AgentX Team
@@ -10,17 +10,17 @@ agentx/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
10
10
  agentx/testing.py,sha256=VKry0lgqaeMyxy9aPuSV73LTUhRv0WcInVVWOdkz4e0,7194
11
11
  agentx/traces.py,sha256=sz9gxutlDKNgf0fdCkyzKsGwK5eMAJuIEl3NIwGrsuo,2287
12
12
  agentx/util.py,sha256=lt2Kpg4Fj8whrSuAhLEahcPEIYRVpK9RCC5y9ShdTJs,703
13
- agentx/version.py,sha256=1VXNrsOvWIerd1gj2Z41YiUd-fiUy2DWAvpw33fR7ig,18
13
+ agentx/version.py,sha256=IExNVK5GJVnQo4b-FGsI7pTg9u4gzQTgCzUF5lFpZ_I,18
14
14
  agentx/evaluations/__init__.py,sha256=Erv7RGFlRxGTG4rVb2uHhCqLLEX_iAWKqkZKzK6CumE,262
15
15
  agentx/evaluations/_term.py,sha256=WFpiNzdgDBeJJ-Gg-6X7TwwxllobuE4OqFUTuDQvS3Y,2529
16
16
  agentx/evaluations/client.py,sha256=1MsduXHOSkwS40RvbUpNgGMU4pZvFtChqlU82YqKdOw,30028
17
17
  agentx/evaluations/datasets.py,sha256=7OYj5pXpJenViAhPe7Sr_7Sf9KWd7b1OhV-dPL2tso8,14283
18
18
  agentx/evaluations/evaluation_settings.py,sha256=geRydHqKpxwkxBVfMIOqkB6NlStyOZrYwZnMHluDwvA,6276
19
- agentx/evaluations/models.py,sha256=Q2oV4n5DPlWGY68a1D50AYJuAzBkjCEe1RoyFs4eLhY,31221
19
+ agentx/evaluations/models.py,sha256=b8cp0arikLW6tZqzmRs-4DVqgFs9FSR5DU9LleO82VY,31699
20
20
  agentx/evaluations/prompts.py,sha256=8xyvMpAl3mD5xuSZozuaohdnCBWe5Q02RldwiPLUO20,3493
21
21
  agentx/evaluations/reporting.py,sha256=GtNnL-1eNEQrSqd0yrHi5mGqC9_u0kUAN8Mbjdhw6yY,6389
22
22
  agentx/evaluations/results.py,sha256=w7TRUik4eXIqrznXvlF_L9hnObnHvgdRKvihfN1XeAU,4692
23
- agentx/evaluations/runner.py,sha256=jPUPpuziRLUc8m_wbWwUvUrdwfSZ0cMRRnTdUSY-qsY,27332
23
+ agentx/evaluations/runner.py,sha256=6k-fUYAwf_sbxSWlirQzI7D60ODjSfvGUKeLXnSKD_Q,27947
24
24
  agentx/evaluations/tool_schemas.py,sha256=ZyrnSOnx9xlSAj4swfDey9ZLS2ly-g21DByTwRAGwNA,2323
25
25
  agentx/evaluations/tracing.py,sha256=MSJD9bzfInMi7E70mc8mWUc0fx5DLkOgUzRru2mNQoc,1825
26
26
  agentx/evaluations/adapters/__init__.py,sha256=fK8Hx75usbiY03XUSmnLnSrwBXip2CQs24YZcdEgj6w,287
@@ -58,11 +58,12 @@ agentx/resources/conversation.py,sha256=94pOdM6Pmj67RhikZVBNGkT3V-WKK_ZAD7s27p5-
58
58
  agentx/resources/workforce.py,sha256=lfGVkoV9lcOp-lZScjTvwRarJMkhzfMfLm4syJDujGc,4168
59
59
  agentx/tracing/__init__.py,sha256=l2mIRILE-wWrCD0fekgrIOQUEvtDObzdrkgpQUtXroU,408
60
60
  agentx/tracing/ci_types.py,sha256=zFVcZGvc1qbVDdOlEPQ1i5R6terqJHxCzFF4ZjouxxI,1423
61
+ agentx/tracing/eval_scope.py,sha256=ElMbPxpqpQBVuaUnHNlw9RIQu71oyOpd0yR55bDH8eI,2144
61
62
  agentx/tracing/ingest_client.py,sha256=5X6RyxtMq1XYgOzsMd3b2f2qpmYpsMnhKm7NB3oup00,17664
62
- agentx/tracing/tracer.py,sha256=iZymguMO3TOoS1YDbS6RfEdt7CWWavA3HO4iiSnnVbw,52893
63
- agentx_python-0.8.2.dist-info/licenses/LICENSE,sha256=gZVsM-nLsE8vlaY6NXXsVoo6IlCClxkToAWmhgT3y_s,10762
64
- agentx_python-0.8.2.dist-info/METADATA,sha256=eoN7qIXyCrvWO_HYlipOJ-bAGFGxOSh_y8mG46x81hw,20992
65
- agentx_python-0.8.2.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
66
- agentx_python-0.8.2.dist-info/entry_points.txt,sha256=rQqF1JTY3T1yfviBU1r8PU-5mBdfOk6P4bZ1L4BskIM,172
67
- agentx_python-0.8.2.dist-info/top_level.txt,sha256=s-q-HB9Gb_QdrZNacSeQyF_c25gQooMy7DlxzgLOHPk,7
68
- agentx_python-0.8.2.dist-info/RECORD,,
63
+ agentx/tracing/tracer.py,sha256=qQojU-T4O8IJKyf9vue0Em66qPBTZKl3FMyr3KJKPnI,53770
64
+ agentx_python-0.8.4.dist-info/licenses/LICENSE,sha256=gZVsM-nLsE8vlaY6NXXsVoo6IlCClxkToAWmhgT3y_s,10762
65
+ agentx_python-0.8.4.dist-info/METADATA,sha256=g4P1EOltG3-fpNZuo9lmVGA3UhcFuxnYQxj1l-57X44,20992
66
+ agentx_python-0.8.4.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
67
+ agentx_python-0.8.4.dist-info/entry_points.txt,sha256=rQqF1JTY3T1yfviBU1r8PU-5mBdfOk6P4bZ1L4BskIM,172
68
+ agentx_python-0.8.4.dist-info/top_level.txt,sha256=s-q-HB9Gb_QdrZNacSeQyF_c25gQooMy7DlxzgLOHPk,7
69
+ agentx_python-0.8.4.dist-info/RECORD,,