agentx-python 0.1__py3-none-any.whl → 0.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentx/__init__.py +31 -0
- agentx/agentx.py +195 -0
- agentx/cli.py +158 -0
- agentx/evaluations/__init__.py +7 -0
- agentx/evaluations/_term.py +102 -0
- agentx/evaluations/adapters/__init__.py +5 -0
- agentx/evaluations/adapters/http_endpoint.py +69 -0
- agentx/evaluations/adapters/precomputed.py +35 -0
- agentx/evaluations/adapters/raw.py +38 -0
- agentx/evaluations/client.py +636 -0
- agentx/evaluations/datasets.py +347 -0
- agentx/evaluations/evaluation_settings.py +146 -0
- agentx/evaluations/models.py +744 -0
- agentx/evaluations/prompts.py +74 -0
- agentx/evaluations/reporting.py +186 -0
- agentx/evaluations/results.py +137 -0
- agentx/evaluations/runner.py +639 -0
- agentx/evaluations/tool_schemas.py +48 -0
- agentx/evaluations/tracing.py +54 -0
- agentx/exceptions.py +50 -0
- agentx/export.py +99 -0
- agentx/feedback.py +94 -0
- agentx/integrations/__init__.py +10 -0
- agentx/integrations/_traced_call.py +168 -0
- agentx/integrations/anthropic.py +285 -0
- agentx/integrations/autogen.py +200 -0
- agentx/integrations/crewai.py +251 -0
- agentx/integrations/databricks.py +405 -0
- agentx/integrations/google_adk.py +315 -0
- agentx/integrations/google_genai.py +329 -0
- agentx/integrations/langchain.py +940 -0
- agentx/integrations/litellm.py +154 -0
- agentx/integrations/llamaindex.py +303 -0
- agentx/integrations/moveworks.py +587 -0
- agentx/integrations/openai.py +169 -0
- agentx/integrations/openai_agents.py +309 -0
- agentx/monitor/__init__.py +17 -0
- agentx/monitor/agents.py +28 -0
- agentx/monitor/client.py +365 -0
- agentx/monitor/judge_scorers.py +334 -0
- agentx/monitor/models.py +184 -0
- agentx/monitor/online_evaluators.py +173 -0
- agentx/monitor/patterns.py +121 -0
- agentx/monitor/profile.py +72 -0
- agentx/monitor/scorers.py +172 -0
- agentx/monitor/sessions.py +21 -0
- agentx/monitor/signals.py +41 -0
- agentx/outcomes.py +86 -0
- agentx/projects.py +61 -0
- agentx/py.typed +0 -0
- agentx/resources/__init__.py +0 -0
- agentx/resources/agent.py +56 -0
- agentx/resources/conversation.py +128 -0
- agentx/resources/workforce.py +121 -0
- agentx/testing.py +90 -0
- agentx/traces.py +64 -0
- agentx/tracing/__init__.py +21 -0
- agentx/tracing/ci_types.py +66 -0
- agentx/tracing/ingest_client.py +440 -0
- agentx/tracing/tracer.py +1165 -0
- agentx/util.py +20 -0
- agentx/version.py +1 -0
- agentx_python-0.1.1.dist-info/METADATA +422 -0
- agentx_python-0.1.1.dist-info/RECORD +68 -0
- {agentx_python-0.1.dist-info → agentx_python-0.1.1.dist-info}/WHEEL +1 -1
- agentx_python-0.1.1.dist-info/entry_points.txt +4 -0
- agentx_python-0.1.1.dist-info/licenses/LICENSE +190 -0
- agentx_python-0.1.1.dist-info/top_level.txt +1 -0
- agentx_python/__init__.py +0 -19
- agentx_python/agent.py +0 -55
- agentx_python/version.py +0 -1
- agentx_python-0.1.dist-info/LICENSE +0 -21
- agentx_python-0.1.dist-info/METADATA +0 -32
- agentx_python-0.1.dist-info/RECORD +0 -8
- agentx_python-0.1.dist-info/top_level.txt +0 -1
agentx/__init__.py
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
|
|
3
|
+
from agentx.agentx import AgentX
|
|
4
|
+
from agentx.version import VERSION
|
|
5
|
+
from agentx.exceptions import (
|
|
6
|
+
AgentXError,
|
|
7
|
+
AgentXAuthError,
|
|
8
|
+
AgentXAPIError,
|
|
9
|
+
DatasetNotFound,
|
|
10
|
+
CINotEnabled,
|
|
11
|
+
CIRunExpired,
|
|
12
|
+
CIGateFailure,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
logging.basicConfig(
|
|
16
|
+
level=logging.INFO,
|
|
17
|
+
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
|
|
18
|
+
datefmt="%Y-%m-%d %H:%M:%S %Z",
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
__all__ = [
|
|
22
|
+
"AgentX",
|
|
23
|
+
"AgentXError",
|
|
24
|
+
"AgentXAuthError",
|
|
25
|
+
"AgentXAPIError",
|
|
26
|
+
"DatasetNotFound",
|
|
27
|
+
"CINotEnabled",
|
|
28
|
+
"CIRunExpired",
|
|
29
|
+
"CIGateFailure",
|
|
30
|
+
]
|
|
31
|
+
__version__ = VERSION
|
agentx/agentx.py
ADDED
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
from typing import List, Optional
|
|
2
|
+
import requests
|
|
3
|
+
import os
|
|
4
|
+
import logging
|
|
5
|
+
|
|
6
|
+
from agentx.util import get_headers, api_base
|
|
7
|
+
from agentx.resources.agent import Agent
|
|
8
|
+
from agentx.resources.workforce import Workforce
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class AgentX:
|
|
12
|
+
|
|
13
|
+
def __init__(
|
|
14
|
+
self,
|
|
15
|
+
api_key: Optional[str] = None,
|
|
16
|
+
base_url: Optional[str] = None,
|
|
17
|
+
workspace_id: Optional[str] = None,
|
|
18
|
+
):
|
|
19
|
+
self.api_key = api_key or os.getenv("AGENTX_API_KEY")
|
|
20
|
+
if self.api_key and not os.getenv("AGENTX_API_KEY"):
|
|
21
|
+
os.environ["AGENTX_API_KEY"] = self.api_key
|
|
22
|
+
|
|
23
|
+
# base_url overrides AGENTX_API_BASE_URL env var (and the SDK default). It is
|
|
24
|
+
# deliberately NOT written back into os.environ: the constructor used to do that, which
|
|
25
|
+
# made the last-constructed client silently re-point every other client in the process
|
|
26
|
+
# (deep-dive round 3, bug #1). Each sub-client below receives this value explicitly and
|
|
27
|
+
# captures it at construction instead.
|
|
28
|
+
self.base_url = base_url or os.getenv("AGENTX_API_BASE_URL")
|
|
29
|
+
|
|
30
|
+
self.workspace_id = workspace_id or os.getenv("AGENTX_WORKSPACE_ID")
|
|
31
|
+
|
|
32
|
+
from agentx.evaluations.client import EvaluationsClient
|
|
33
|
+
from agentx.evaluations.runner import EvaluationsRunner
|
|
34
|
+
from agentx.monitor.client import MonitorClient
|
|
35
|
+
from agentx.tracing.ingest_client import IngestClient
|
|
36
|
+
from agentx.tracing.tracer import Tracer
|
|
37
|
+
from agentx.version import VERSION
|
|
38
|
+
|
|
39
|
+
_eval_client = EvaluationsClient(
|
|
40
|
+
api_key=self.api_key,
|
|
41
|
+
sdk_version=VERSION,
|
|
42
|
+
base_url=self.base_url,
|
|
43
|
+
workspace_id=self.workspace_id,
|
|
44
|
+
)
|
|
45
|
+
self.evaluations = EvaluationsRunner(_eval_client)
|
|
46
|
+
|
|
47
|
+
# Monitor: create/reuse patterns (client.monitor.patterns) that a trace can be checked
|
|
48
|
+
# against at send time via tracer.trace(..., monitor=True, pattern_ids=[...]), then read
|
|
49
|
+
# back the resulting alerts/findings with client.monitor.signals. Per-agent coverage and
|
|
50
|
+
# detection settings (sample rate, retention, threshold overrides like the built-in
|
|
51
|
+
# "Latency regression" pattern's threshold) are client.monitor.profile.
|
|
52
|
+
self.monitor = MonitorClient(
|
|
53
|
+
api_key=self.api_key,
|
|
54
|
+
sdk_version=VERSION,
|
|
55
|
+
base_url=self.base_url,
|
|
56
|
+
workspace_id=self.workspace_id,
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
from agentx.outcomes import OutcomesClient
|
|
60
|
+
|
|
61
|
+
# Report real, after-the-fact outcomes ("the ticket got reopened") against traces - the
|
|
62
|
+
# ground truth behind the dashboard's Judge Calibration card. Self-host only.
|
|
63
|
+
self.outcomes = OutcomesClient(api_key=self.api_key, base_url=self.base_url)
|
|
64
|
+
|
|
65
|
+
from agentx.projects import ProjectsClient
|
|
66
|
+
|
|
67
|
+
# Project CRUD (self-host): isolated tenants with their own API keys (P1.1).
|
|
68
|
+
self.projects = ProjectsClient(api_key=self.api_key, base_url=self.base_url)
|
|
69
|
+
|
|
70
|
+
from agentx.traces import TracesClient
|
|
71
|
+
|
|
72
|
+
# The read side of tracing: trace-by-id detail and paginated listing (P1.2).
|
|
73
|
+
self.traces = TracesClient(api_key=self.api_key, base_url=self.base_url)
|
|
74
|
+
|
|
75
|
+
from agentx.export import ExportClient
|
|
76
|
+
|
|
77
|
+
# Bulk NDJSON egress for backup/migration (P2.1): manifest, per-entity streaming, and
|
|
78
|
+
# directory dumps. Self-host only.
|
|
79
|
+
self.export = ExportClient(api_key=self.api_key, base_url=self.base_url)
|
|
80
|
+
|
|
81
|
+
from agentx.feedback import FeedbackClient
|
|
82
|
+
|
|
83
|
+
# Forward end-user votes ("up"/"down") on traced responses - a "down" raises a signal
|
|
84
|
+
# directly, and every vote feeds Judge Calibration alongside outcomes. Self-host only.
|
|
85
|
+
self.feedback = FeedbackClient(api_key=self.api_key, base_url=self.base_url)
|
|
86
|
+
|
|
87
|
+
_ingest_client = IngestClient(
|
|
88
|
+
api_key=self.api_key,
|
|
89
|
+
sdk_version=VERSION,
|
|
90
|
+
base_url=self.base_url,
|
|
91
|
+
workspace_id=self.workspace_id,
|
|
92
|
+
)
|
|
93
|
+
self.tracer = Tracer(_ingest_client)
|
|
94
|
+
|
|
95
|
+
@classmethod
|
|
96
|
+
def from_env(cls) -> "AgentX":
|
|
97
|
+
"""Create an AgentX client from the environment: AGENTX_API_KEY plus, for the base URL,
|
|
98
|
+
the first of AGENTX_API_BASE_URL / AGENTX_SELFHOST_BASE_URL / BASE_URL that is set.
|
|
99
|
+
The fallbacks match the conventions the self-host samples and .env files already use,
|
|
100
|
+
so from_env works wherever an explicit AgentX(base_url=...) would."""
|
|
101
|
+
import os
|
|
102
|
+
|
|
103
|
+
base_url = (
|
|
104
|
+
os.getenv("AGENTX_API_BASE_URL")
|
|
105
|
+
or os.getenv("AGENTX_SELFHOST_BASE_URL")
|
|
106
|
+
or os.getenv("BASE_URL")
|
|
107
|
+
)
|
|
108
|
+
return cls(base_url=base_url) if base_url else cls()
|
|
109
|
+
|
|
110
|
+
def get_agent(self, id: str) -> Agent:
|
|
111
|
+
url = f"{self.base_url or api_base()}/access/agents/{id}"
|
|
112
|
+
# Make a GET request to the AgentX API
|
|
113
|
+
response = requests.get(url, headers=get_headers(self.api_key))
|
|
114
|
+
# Check if response was successful
|
|
115
|
+
if response.status_code == 200:
|
|
116
|
+
return Agent(**response.json())
|
|
117
|
+
else:
|
|
118
|
+
raise Exception(f"Failed to retrieve agent: {response.reason}")
|
|
119
|
+
|
|
120
|
+
def list_agents(self) -> List[Agent]:
|
|
121
|
+
url = f"{self.base_url or api_base()}/access/agents"
|
|
122
|
+
# Make a GET request to the AgentX API
|
|
123
|
+
response = requests.get(url, headers=get_headers(self.api_key))
|
|
124
|
+
# Check if response was successful
|
|
125
|
+
if response.status_code == 200:
|
|
126
|
+
return [Agent(**agent) for agent in response.json()]
|
|
127
|
+
else:
|
|
128
|
+
raise Exception(f"Failed to list agents: {response.reason}")
|
|
129
|
+
|
|
130
|
+
@staticmethod
|
|
131
|
+
def list_workforces() -> List["Workforce"]:
|
|
132
|
+
"""List all workforces/teams."""
|
|
133
|
+
url = f"{api_base()}/access/teams"
|
|
134
|
+
response = requests.get(url, headers=get_headers())
|
|
135
|
+
if response.status_code == 200:
|
|
136
|
+
return [Workforce(**workforce) for workforce in response.json()]
|
|
137
|
+
else:
|
|
138
|
+
raise Exception(
|
|
139
|
+
f"Failed to list workforces: {response.status_code} - {response.reason}"
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
def ping(self) -> dict:
|
|
143
|
+
"""Verify the client can actually reach AgentX and that the API key is accepted.
|
|
144
|
+
|
|
145
|
+
The constructor is deliberately lazy (no network call - standard SDK behavior, so
|
|
146
|
+
offline construction and tests work), and trace delivery is fire-and-forget, so a
|
|
147
|
+
wrong ``base_url`` or ``api_key`` otherwise surfaces only as a one-time warning in
|
|
148
|
+
logs while traces silently go nowhere. Call this once at startup of a long-running
|
|
149
|
+
service to fail fast instead::
|
|
150
|
+
|
|
151
|
+
client = AgentX.from_env()
|
|
152
|
+
client.ping() # raises immediately on a bad URL or key
|
|
153
|
+
|
|
154
|
+
Raises :class:`agentx.exceptions.AgentXConnectionError` when the URL is unreachable,
|
|
155
|
+
:class:`agentx.exceptions.AgentXAuthError` when the key is rejected, and
|
|
156
|
+
:class:`agentx.exceptions.AgentXAPIError` on any other non-OK response. Returns
|
|
157
|
+
``{"ok": True, "base_url": ...}`` on success.
|
|
158
|
+
"""
|
|
159
|
+
from agentx.exceptions import AgentXAPIError, AgentXAuthError, AgentXConnectionError
|
|
160
|
+
|
|
161
|
+
base = (self.base_url or api_base()).rstrip("/")
|
|
162
|
+
# /monitor/patterns: the cheapest key-authenticated endpoint that exists on both the
|
|
163
|
+
# hosted API and the self-host engine's SDK-facing router.
|
|
164
|
+
url = f"{base}/monitor/patterns"
|
|
165
|
+
try:
|
|
166
|
+
response = requests.get(url, headers=get_headers(self.api_key), timeout=10)
|
|
167
|
+
except requests.RequestException as exc:
|
|
168
|
+
raise AgentXConnectionError(
|
|
169
|
+
f"Cannot reach AgentX at {base} ({exc.__class__.__name__}: {exc}). "
|
|
170
|
+
"Check base_url / AGENTX_API_BASE_URL - for self-host it should look like "
|
|
171
|
+
"http://localhost:4700/api/v1."
|
|
172
|
+
) from exc
|
|
173
|
+
if response.status_code in (401, 403):
|
|
174
|
+
raise AgentXAuthError(
|
|
175
|
+
f"AgentX at {base} rejected the API key (HTTP {response.status_code}). "
|
|
176
|
+
"Check api_key / AGENTX_API_KEY - for self-host, copy the 'Default project "
|
|
177
|
+
"API key' from the engine's startup log."
|
|
178
|
+
)
|
|
179
|
+
if not response.ok:
|
|
180
|
+
raise AgentXAPIError(
|
|
181
|
+
f"AgentX at {base} responded HTTP {response.status_code} to the health probe.",
|
|
182
|
+
status_code=response.status_code,
|
|
183
|
+
)
|
|
184
|
+
return {"ok": True, "base_url": base}
|
|
185
|
+
|
|
186
|
+
def get_profile(self):
|
|
187
|
+
"""Get the current user's profile information."""
|
|
188
|
+
url = f"{self.base_url or api_base()}/access/getProfile"
|
|
189
|
+
response = requests.get(url, headers=get_headers(self.api_key))
|
|
190
|
+
if response.status_code == 200:
|
|
191
|
+
return response.json()
|
|
192
|
+
else:
|
|
193
|
+
raise Exception(
|
|
194
|
+
f"Failed to get profile: {response.status_code} - {response.reason}"
|
|
195
|
+
)
|
agentx/cli.py
ADDED
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
"""
|
|
2
|
+
`agentx-trace-eval` - thin launcher for AgentX's self-hostable governance engine (Trace,
|
|
3
|
+
Evaluate, Monitor), published separately at github.com/AgentX-ai/AgentX-trace-eval (a Go CLI
|
|
4
|
+
wrapping a Bun-compiled TypeScript engine, not Python). That compiled engine binary is tens of
|
|
5
|
+
megabytes; most `pip install agentx-python` installs are just this SDK talking to the hosted
|
|
6
|
+
AgentX SaaS and would never touch it, so it isn't bundled in this package. Instead, this command
|
|
7
|
+
downloads the matching release into ~/.agentx/bin the first time it's needed (mirroring
|
|
8
|
+
AgentX-trace-eval's own install.sh) and then hands off to the real `agentx-server` binary.
|
|
9
|
+
|
|
10
|
+
Usage:
|
|
11
|
+
agentx-trace-eval --dev
|
|
12
|
+
agentx-trace-eval --port 5000 --db-url postgres://...
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import os
|
|
16
|
+
import platform
|
|
17
|
+
import shutil
|
|
18
|
+
import stat
|
|
19
|
+
import sys
|
|
20
|
+
import tarfile
|
|
21
|
+
import tempfile
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import Tuple
|
|
24
|
+
|
|
25
|
+
import requests
|
|
26
|
+
|
|
27
|
+
REPO = "AgentX-ai/AgentX-trace-eval"
|
|
28
|
+
INSTALL_DIR = Path(os.environ.get("AGENTX_INSTALL_DIR", str(Path.home() / ".agentx" / "bin")))
|
|
29
|
+
_BIN_NAMES = ("agentx", "agentx-server", "agentx-engine")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _platform_tag() -> Tuple[str, str]:
|
|
33
|
+
system = platform.system()
|
|
34
|
+
if system == "Darwin":
|
|
35
|
+
os_name = "darwin"
|
|
36
|
+
elif system == "Linux":
|
|
37
|
+
os_name = "linux"
|
|
38
|
+
else:
|
|
39
|
+
raise SystemExit(
|
|
40
|
+
f"agentx-trace-eval: unsupported OS {system!r} (self-host currently supports macOS and Linux)"
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
machine = platform.machine().lower()
|
|
44
|
+
if machine in ("arm64", "aarch64"):
|
|
45
|
+
arch = "arm64"
|
|
46
|
+
elif machine in ("x86_64", "amd64"):
|
|
47
|
+
arch = "amd64"
|
|
48
|
+
else:
|
|
49
|
+
raise SystemExit(f"agentx-trace-eval: unsupported architecture {machine!r}")
|
|
50
|
+
|
|
51
|
+
return os_name, arch
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _release_url(asset: str, version: str) -> str:
|
|
55
|
+
if version == "latest":
|
|
56
|
+
return f"https://github.com/{REPO}/releases/latest/download/{asset}"
|
|
57
|
+
return f"https://github.com/{REPO}/releases/download/{version}/{asset}"
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _download(url: str, dest: Path) -> None:
|
|
61
|
+
response = requests.get(url, stream=True, timeout=60)
|
|
62
|
+
response.raise_for_status()
|
|
63
|
+
with open(dest, "wb") as f:
|
|
64
|
+
for chunk in response.iter_content(chunk_size=1 << 16):
|
|
65
|
+
f.write(chunk)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _extract_tar(archive: Path, dest_dir: Path) -> None:
|
|
69
|
+
with tarfile.open(archive) as tar:
|
|
70
|
+
try:
|
|
71
|
+
# filter="data" (PEP 706, Python 3.12+) rejects absolute paths/symlink escapes.
|
|
72
|
+
# Belt-and-suspenders here since these are trusted release assets built by our own CI
|
|
73
|
+
# (see REPO above), not arbitrary user-supplied archives.
|
|
74
|
+
tar.extractall(dest_dir, filter="data")
|
|
75
|
+
except TypeError:
|
|
76
|
+
tar.extractall(dest_dir) # Python < 3.12: filter kwarg doesn't exist yet
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _install(version: str = "latest") -> None:
|
|
80
|
+
os_name, arch = _platform_tag()
|
|
81
|
+
INSTALL_DIR.mkdir(parents=True, exist_ok=True)
|
|
82
|
+
|
|
83
|
+
print(f"agentx-trace-eval: downloading agentx ({os_name}/{arch})...", file=sys.stderr)
|
|
84
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
85
|
+
tmp_dir = Path(tmp)
|
|
86
|
+
archive = tmp_dir / "agentx.tar.gz"
|
|
87
|
+
url = _release_url(f"agentx_{os_name}_{arch}.tar.gz", version)
|
|
88
|
+
try:
|
|
89
|
+
_download(url, archive)
|
|
90
|
+
except requests.HTTPError as exc:
|
|
91
|
+
raise SystemExit(
|
|
92
|
+
f"agentx-trace-eval: failed to download {url} ({exc}).\n"
|
|
93
|
+
"No published release found; see AgentX-trace-eval's README for building from source."
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
_extract_tar(archive, tmp_dir)
|
|
97
|
+
|
|
98
|
+
found_any = False
|
|
99
|
+
for name in _BIN_NAMES:
|
|
100
|
+
src = tmp_dir / name
|
|
101
|
+
if not src.exists():
|
|
102
|
+
continue
|
|
103
|
+
found_any = True
|
|
104
|
+
dest = INSTALL_DIR / name
|
|
105
|
+
shutil.move(str(src), str(dest))
|
|
106
|
+
dest.chmod(dest.stat().st_mode | stat.S_IEXEC | stat.S_IXGRP | stat.S_IXOTH)
|
|
107
|
+
|
|
108
|
+
if not found_any:
|
|
109
|
+
raise SystemExit(f"agentx-trace-eval: downloaded archive from {url} didn't contain any of {_BIN_NAMES}")
|
|
110
|
+
|
|
111
|
+
if os.environ.get("AGENTX_TRACE_EVAL_SKIP_WEB"):
|
|
112
|
+
return
|
|
113
|
+
|
|
114
|
+
# Best-effort: the dashboard is a separate, platform-independent asset (see
|
|
115
|
+
# AgentX-trace-eval's README's "Dashboard release"). Missing it shouldn't block getting the
|
|
116
|
+
# engine running headless, so a failure here warns and continues rather than raising.
|
|
117
|
+
print("agentx-trace-eval: downloading dashboard...", file=sys.stderr)
|
|
118
|
+
web_dir = INSTALL_DIR / "web"
|
|
119
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
120
|
+
web_archive = Path(tmp) / "agentx-web.tar.gz"
|
|
121
|
+
web_url = _release_url("agentx-web.tar.gz", version)
|
|
122
|
+
try:
|
|
123
|
+
_download(web_url, web_archive)
|
|
124
|
+
except requests.HTTPError:
|
|
125
|
+
print(
|
|
126
|
+
f"agentx-trace-eval: no dashboard bundle found at {web_url}, continuing without one",
|
|
127
|
+
file=sys.stderr,
|
|
128
|
+
)
|
|
129
|
+
return
|
|
130
|
+
if web_dir.exists():
|
|
131
|
+
shutil.rmtree(web_dir)
|
|
132
|
+
web_dir.mkdir(parents=True)
|
|
133
|
+
_extract_tar(web_archive, web_dir)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def ensure_installed(version: str = "latest") -> Path:
|
|
137
|
+
"""Downloads agentx-server (+ its engine) into ~/.agentx/bin if not already present there.
|
|
138
|
+
Returns the path to the agentx-server executable. Set AGENTX_INSTALL_DIR to change where
|
|
139
|
+
this looks/installs; set AGENTX_TRACE_EVAL_VERSION to pin a release tag instead of latest."""
|
|
140
|
+
server_path = INSTALL_DIR / "agentx-server"
|
|
141
|
+
if not server_path.exists():
|
|
142
|
+
_install(version=version)
|
|
143
|
+
return server_path
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def main() -> None:
|
|
147
|
+
version = os.environ.get("AGENTX_TRACE_EVAL_VERSION", "latest")
|
|
148
|
+
server_path = ensure_installed(version=version)
|
|
149
|
+
if not server_path.exists():
|
|
150
|
+
raise SystemExit(f"agentx-trace-eval: {server_path} still missing after install, giving up")
|
|
151
|
+
|
|
152
|
+
# os.execv replaces this process rather than spawning a subprocess: signals, stdio, and the
|
|
153
|
+
# exit code all pass straight through to agentx-server, same as invoking it directly.
|
|
154
|
+
os.execv(str(server_path), [str(server_path)] + sys.argv[1:])
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
if __name__ == "__main__":
|
|
158
|
+
main()
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""Minimal ANSI terminal helpers - no external dependencies."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import itertools
|
|
6
|
+
import sys
|
|
7
|
+
import os
|
|
8
|
+
import threading
|
|
9
|
+
import time
|
|
10
|
+
|
|
11
|
+
_IS_TTY = hasattr(sys.stdout, "isatty") and sys.stdout.isatty()
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _c(code: str) -> str:
|
|
15
|
+
return f"\033[{code}m" if _IS_TTY else ""
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
RESET = _c("0")
|
|
19
|
+
BOLD = _c("1")
|
|
20
|
+
DIM = _c("2")
|
|
21
|
+
GREEN = _c("32")
|
|
22
|
+
YELLOW = _c("33")
|
|
23
|
+
RED = _c("31")
|
|
24
|
+
CYAN = _c("36")
|
|
25
|
+
MAGENTA = _c("35")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def green(s: str) -> str:
|
|
29
|
+
return f"{GREEN}{s}{RESET}"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def yellow(s: str) -> str:
|
|
33
|
+
return f"{YELLOW}{s}{RESET}"
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def red(s: str) -> str:
|
|
37
|
+
return f"{RED}{s}{RESET}"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def cyan(s: str) -> str:
|
|
41
|
+
return f"{CYAN}{s}{RESET}"
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def bold(s: str) -> str:
|
|
45
|
+
return f"{BOLD}{s}{RESET}"
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def dim(s: str) -> str:
|
|
49
|
+
return f"{DIM}{s}{RESET}"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def magenta(s: str) -> str:
|
|
53
|
+
return f"{MAGENTA}{s}{RESET}"
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class Spinner:
|
|
57
|
+
"""Inline spinner that overwrites the current line on a TTY."""
|
|
58
|
+
|
|
59
|
+
_FRAMES = ("⠋", "⠙", "⠹", "⠸", "⠼", "⠴", "⠦", "⠧", "⠇", "⠏")
|
|
60
|
+
|
|
61
|
+
def __init__(self, message: str):
|
|
62
|
+
self._message = message
|
|
63
|
+
self._stop = threading.Event()
|
|
64
|
+
self._thread: threading.Thread | None = None
|
|
65
|
+
# AGENTX_EVAL_QUIET=1: no spinner thread at all - CI logs stay clean.
|
|
66
|
+
self._quiet = os.getenv("AGENTX_EVAL_QUIET", "").lower() in ("1", "true", "yes")
|
|
67
|
+
|
|
68
|
+
def __enter__(self) -> "Spinner":
|
|
69
|
+
if self._quiet:
|
|
70
|
+
return self
|
|
71
|
+
if not _IS_TTY:
|
|
72
|
+
print(f" {self._message}...", flush=True)
|
|
73
|
+
return self
|
|
74
|
+
self._stop.clear()
|
|
75
|
+
self._thread = threading.Thread(target=self._spin, daemon=True)
|
|
76
|
+
self._thread.start()
|
|
77
|
+
return self
|
|
78
|
+
|
|
79
|
+
def update(self, message: str) -> None:
|
|
80
|
+
"""Change the displayed message while the spinner keeps running."""
|
|
81
|
+
self._message = message
|
|
82
|
+
if not _IS_TTY:
|
|
83
|
+
print(f" {message}...", flush=True)
|
|
84
|
+
|
|
85
|
+
def __exit__(self, *_) -> None:
|
|
86
|
+
if self._quiet or not _IS_TTY:
|
|
87
|
+
return
|
|
88
|
+
self._stop.set()
|
|
89
|
+
if self._thread:
|
|
90
|
+
self._thread.join()
|
|
91
|
+
sys.stdout.write(f"\r{' ' * (len(self._message) + 12)}\r")
|
|
92
|
+
sys.stdout.flush()
|
|
93
|
+
|
|
94
|
+
def _spin(self) -> None:
|
|
95
|
+
for frame in itertools.cycle(self._FRAMES):
|
|
96
|
+
if self._stop.is_set():
|
|
97
|
+
break
|
|
98
|
+
sys.stdout.write(
|
|
99
|
+
f"\r {CYAN}{frame}{RESET} {self._message} {DIM}(this may take ~60s+){RESET}"
|
|
100
|
+
)
|
|
101
|
+
sys.stdout.flush()
|
|
102
|
+
time.sleep(0.08)
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
from agentx.evaluations.adapters.raw import RawCallableAdapter
|
|
2
|
+
from agentx.evaluations.adapters.precomputed import PrecomputedAdapter
|
|
3
|
+
from agentx.evaluations.adapters.http_endpoint import HttpEndpointAdapter
|
|
4
|
+
|
|
5
|
+
__all__ = ["RawCallableAdapter", "PrecomputedAdapter", "HttpEndpointAdapter"]
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import time
|
|
4
|
+
from typing import Any, Dict, Optional
|
|
5
|
+
|
|
6
|
+
import requests
|
|
7
|
+
|
|
8
|
+
from agentx.evaluations.models import EvaluationCase, EvaluationResult
|
|
9
|
+
from agentx.evaluations.results import normalize_error, normalize_result
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class HttpEndpointAdapter:
|
|
13
|
+
"""
|
|
14
|
+
Calls a user-hosted HTTP endpoint for each evaluation case.
|
|
15
|
+
The SDK (running locally) makes the request - the AgentX API never
|
|
16
|
+
touches the customer's endpoint.
|
|
17
|
+
|
|
18
|
+
The endpoint receives a POST with::
|
|
19
|
+
|
|
20
|
+
{"query": "...", "case_id": "...", "metadata": {...}}
|
|
21
|
+
|
|
22
|
+
It should respond with a JSON body containing at least ``output``
|
|
23
|
+
(or ``text``) and optionally ``trace`` and ``metadata``.
|
|
24
|
+
|
|
25
|
+
Usage::
|
|
26
|
+
|
|
27
|
+
adapter = HttpEndpointAdapter(
|
|
28
|
+
url="http://localhost:8080/eval",
|
|
29
|
+
headers={"Authorization": "Bearer my-token"},
|
|
30
|
+
timeout=30,
|
|
31
|
+
)
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
def __init__(
|
|
35
|
+
self,
|
|
36
|
+
url: str,
|
|
37
|
+
headers: Optional[Dict[str, str]] = None,
|
|
38
|
+
timeout: int = 30,
|
|
39
|
+
method: str = "POST",
|
|
40
|
+
):
|
|
41
|
+
self._url = url
|
|
42
|
+
self._headers = headers or {}
|
|
43
|
+
self._timeout = timeout
|
|
44
|
+
self._method = method.upper()
|
|
45
|
+
|
|
46
|
+
def run(self, case: EvaluationCase) -> EvaluationResult:
|
|
47
|
+
payload = {
|
|
48
|
+
"query": case.query,
|
|
49
|
+
"case_id": case.case_id,
|
|
50
|
+
"question_index": case.question_index,
|
|
51
|
+
"run_number": case.run_number,
|
|
52
|
+
}
|
|
53
|
+
start = time.monotonic()
|
|
54
|
+
try:
|
|
55
|
+
resp = requests.request(
|
|
56
|
+
self._method,
|
|
57
|
+
self._url,
|
|
58
|
+
json=payload,
|
|
59
|
+
headers=self._headers,
|
|
60
|
+
timeout=self._timeout,
|
|
61
|
+
)
|
|
62
|
+
resp.raise_for_status()
|
|
63
|
+
elapsed = int((time.monotonic() - start) * 1000)
|
|
64
|
+
raw = resp.json()
|
|
65
|
+
except Exception as exc:
|
|
66
|
+
elapsed = int((time.monotonic() - start) * 1000)
|
|
67
|
+
return normalize_error(case, exc, latency_ms=elapsed)
|
|
68
|
+
|
|
69
|
+
return normalize_result(case, raw, latency_ms=elapsed)
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Any, Dict, List, Union
|
|
4
|
+
|
|
5
|
+
from agentx.evaluations.models import EvaluationCase, EvaluationResult
|
|
6
|
+
from agentx.evaluations.results import normalize_result
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class PrecomputedAdapter:
|
|
10
|
+
"""
|
|
11
|
+
Adapter for pre-computed outputs - useful when you already have agent
|
|
12
|
+
responses and just want AgentX to score them.
|
|
13
|
+
|
|
14
|
+
Accepts a list or dict keyed by case_id::
|
|
15
|
+
|
|
16
|
+
outputs = {
|
|
17
|
+
"case-0": "You can reset your password from account settings.",
|
|
18
|
+
"case-1": {"output": "Contact support@example.com", "metadata": {...}},
|
|
19
|
+
}
|
|
20
|
+
adapter = PrecomputedAdapter(outputs)
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
def __init__(self, outputs: Union[List[Any], Dict[str, Any]]):
|
|
24
|
+
if isinstance(outputs, list):
|
|
25
|
+
self._lookup: Dict[str, Any] = {str(i): v for i, v in enumerate(outputs)}
|
|
26
|
+
else:
|
|
27
|
+
self._lookup = {str(k): v for k, v in outputs.items()}
|
|
28
|
+
|
|
29
|
+
def run(self, case: EvaluationCase) -> EvaluationResult:
|
|
30
|
+
raw = self._lookup.get(case.case_id) or self._lookup.get(
|
|
31
|
+
str(case.question_index)
|
|
32
|
+
)
|
|
33
|
+
if raw is None:
|
|
34
|
+
raw = ""
|
|
35
|
+
return normalize_result(case, raw, latency_ms=0)
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Any, Callable
|
|
4
|
+
|
|
5
|
+
from agentx.evaluations.models import EvaluationCase, EvaluationResult
|
|
6
|
+
from agentx.evaluations.results import normalize_error, normalize_result
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class RawCallableAdapter:
|
|
10
|
+
"""
|
|
11
|
+
Wraps any Python callable that accepts an EvaluationCase and returns
|
|
12
|
+
str | dict | EvaluationResult.
|
|
13
|
+
|
|
14
|
+
Usage::
|
|
15
|
+
|
|
16
|
+
def my_agent(case: EvaluationCase) -> str:
|
|
17
|
+
return my_llm.invoke(case.query)
|
|
18
|
+
|
|
19
|
+
adapter = RawCallableAdapter(my_agent)
|
|
20
|
+
result = adapter.run(case)
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
def __init__(self, fn: Callable[[EvaluationCase], Any]):
|
|
24
|
+
self._fn = fn
|
|
25
|
+
|
|
26
|
+
def run(
|
|
27
|
+
self, case: EvaluationCase, latency_ms: int | None = None
|
|
28
|
+
) -> EvaluationResult:
|
|
29
|
+
import time
|
|
30
|
+
|
|
31
|
+
start = time.monotonic()
|
|
32
|
+
try:
|
|
33
|
+
raw = self._fn(case)
|
|
34
|
+
except Exception as exc:
|
|
35
|
+
elapsed = int((time.monotonic() - start) * 1000)
|
|
36
|
+
return normalize_error(case, exc, latency_ms=elapsed)
|
|
37
|
+
elapsed = int((time.monotonic() - start) * 1000)
|
|
38
|
+
return normalize_result(case, raw, latency_ms=elapsed)
|