veval-sdk 1.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- veval_sdk-1.1.0/.gitignore +9 -0
- veval_sdk-1.1.0/LICENSE +21 -0
- veval_sdk-1.1.0/PKG-INFO +208 -0
- veval_sdk-1.1.0/README.md +185 -0
- veval_sdk-1.1.0/pyproject.toml +33 -0
- veval_sdk-1.1.0/tests/test_sdk.py +218 -0
- veval_sdk-1.1.0/tests/test_snapshots.py +154 -0
- veval_sdk-1.1.0/veval/__init__.py +47 -0
- veval_sdk-1.1.0/veval/_assertions.py +141 -0
- veval_sdk-1.1.0/veval/_context.py +104 -0
- veval_sdk-1.1.0/veval/_http_client.py +121 -0
- veval_sdk-1.1.0/veval/_options.py +29 -0
- veval_sdk-1.1.0/veval/_scenarios.py +43 -0
- veval_sdk-1.1.0/veval/_sdk.py +332 -0
- veval_sdk-1.1.0/veval/_snapshots.py +431 -0
- veval_sdk-1.1.0/veval/_step.py +68 -0
- veval_sdk-1.1.0/veval/_test_sdk.py +216 -0
- veval_sdk-1.1.0/veval/_tracing.py +121 -0
veval_sdk-1.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Jonathan Dinelle
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
veval_sdk-1.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: veval-sdk
|
|
3
|
+
Version: 1.1.0
|
|
4
|
+
Summary: Trace, evaluate, and test AI agents with a few lines of code.
|
|
5
|
+
Project-URL: Homepage, https://veval.dev
|
|
6
|
+
Project-URL: Documentation, https://docs.veval.dev
|
|
7
|
+
Project-URL: Repository, https://github.com/JonathanDinelle/veval-sdks
|
|
8
|
+
License: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: agents,ai,evaluation,llm,testing,tracing
|
|
11
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
20
|
+
Classifier: Topic :: Software Development :: Testing
|
|
21
|
+
Requires-Python: >=3.10
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
|
|
24
|
+
# veval-sdk
|
|
25
|
+
|
|
26
|
+
Python SDK for [Veval](https://veval.dev) — trace, evaluate, and test AI agents.
|
|
27
|
+
|
|
28
|
+
## Install
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
pip install veval-sdk
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
## Quick start
|
|
35
|
+
|
|
36
|
+
Wrap your agent with `run_async` to send a trace to the Veval dashboard:
|
|
37
|
+
|
|
38
|
+
```python
|
|
39
|
+
import asyncio
|
|
40
|
+
from veval import VevalSdk, VevalOptions, VevalExecutionContext
|
|
41
|
+
|
|
42
|
+
sdk = VevalSdk(VevalOptions(api_key="veval-..."))
|
|
43
|
+
|
|
44
|
+
async def my_agent(ctx: VevalExecutionContext) -> str:
|
|
45
|
+
result = await ctx.track_step_async(
|
|
46
|
+
name="call-llm",
|
|
47
|
+
input=ctx.input,
|
|
48
|
+
step=lambda: call_my_llm(ctx.input),
|
|
49
|
+
)
|
|
50
|
+
return result
|
|
51
|
+
|
|
52
|
+
output = asyncio.run(sdk.run_async("my-agent", my_agent, input="Hello"))
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
Every call to `track_step_async` records the step name, input, output, timing, and any metadata you attach.
|
|
56
|
+
|
|
57
|
+
## Tracking steps
|
|
58
|
+
|
|
59
|
+
Use the `StepHandle` to attach token counts, cost, and model name:
|
|
60
|
+
|
|
61
|
+
```python
|
|
62
|
+
async def agent(ctx: VevalExecutionContext) -> str:
|
|
63
|
+
async def llm_call(handle):
|
|
64
|
+
response = await call_my_llm(ctx.input)
|
|
65
|
+
handle.set_meta("tokens_in", response.usage.input_tokens)
|
|
66
|
+
handle.set_meta("tokens_out", response.usage.output_tokens)
|
|
67
|
+
handle.set_meta("cost_usd", response.usage.cost)
|
|
68
|
+
handle.set_meta("model", "claude-opus-4-6")
|
|
69
|
+
return response.text
|
|
70
|
+
|
|
71
|
+
return await ctx.track_step_async("call-llm", ctx.input, llm_call)
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
## Assertions
|
|
75
|
+
|
|
76
|
+
Use `TraceAssert` to validate behaviour at the trace level:
|
|
77
|
+
|
|
78
|
+
```python
|
|
79
|
+
from veval import TraceAssert
|
|
80
|
+
|
|
81
|
+
assertions = [
|
|
82
|
+
TraceAssert.no_errors(),
|
|
83
|
+
TraceAssert.max_steps(10),
|
|
84
|
+
TraceAssert.step_exists("call-llm"),
|
|
85
|
+
TraceAssert.max_cost(0.05),
|
|
86
|
+
TraceAssert.max_duration(30_000),
|
|
87
|
+
TraceAssert.output_contains("success"),
|
|
88
|
+
TraceAssert.tool_called("web-search"),
|
|
89
|
+
]
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
## Scenarios
|
|
93
|
+
|
|
94
|
+
Run a named scenario against a list of test items and assertions:
|
|
95
|
+
|
|
96
|
+
```python
|
|
97
|
+
from veval import ScenarioItem
|
|
98
|
+
|
|
99
|
+
items = [
|
|
100
|
+
ScenarioItem(name="basic greeting", input="Hello"),
|
|
101
|
+
ScenarioItem(name="edge case", input=""),
|
|
102
|
+
]
|
|
103
|
+
|
|
104
|
+
result = asyncio.run(
|
|
105
|
+
sdk.run_scenario_async("smoke-test", my_agent, assertions, items=items)
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
print(f"Passed: {result.pass_count}/{len(result.results)}")
|
|
109
|
+
for r in result.results:
|
|
110
|
+
if not r.passed:
|
|
111
|
+
print(f" FAIL {r.item.name}: {r.failures}")
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
You can also pull items from the Veval dashboard by omitting `items`:
|
|
115
|
+
|
|
116
|
+
```python
|
|
117
|
+
result = asyncio.run(sdk.run_scenario_async("smoke-test", my_agent, assertions))
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
## Replay and snapshot testing
|
|
121
|
+
|
|
122
|
+
Load a previously recorded trace and replay it with mocked LLM responses — no real calls, deterministic results:
|
|
123
|
+
|
|
124
|
+
```python
|
|
125
|
+
trace = asyncio.run(sdk.get_trace_async("tr_abc123"))
|
|
126
|
+
|
|
127
|
+
replay = asyncio.run(sdk.replay_async(
|
|
128
|
+
trace,
|
|
129
|
+
my_agent,
|
|
130
|
+
ReplayOptions(mock_llm_responses=True, assertions=assertions),
|
|
131
|
+
))
|
|
132
|
+
|
|
133
|
+
print("Passed" if replay.passed else replay.failures)
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
Store a known-good run — every step with its input and output — and detect when a new run drifts from it: a changed prompt or tool argument, a step added, dropped, repeated, or reordered:
|
|
137
|
+
|
|
138
|
+
```python
|
|
139
|
+
# Save a baseline from a recorded trace. The trace is pinned, so retention never deletes it.
|
|
140
|
+
asyncio.run(sdk.save_snapshot_async("my-baseline", "tr_abc123"))
|
|
141
|
+
|
|
142
|
+
# In tests: an assertion like any other. A missing baseline fails; it never passes silently.
|
|
143
|
+
replay = asyncio.run(sdk.replay_async(trace, my_agent, ReplayOptions(
|
|
144
|
+
mock_llm_responses=True,
|
|
145
|
+
assertions=[TraceAssert.matches_snapshot(sdk, "my-baseline")],
|
|
146
|
+
compare_with_recording=SnapshotOptions(), # also fail if steps/inputs drift from the trace
|
|
147
|
+
)))
|
|
148
|
+
|
|
149
|
+
# Or compare directly, and record the result in the dashboard.
|
|
150
|
+
baseline = asyncio.run(sdk.get_snapshot_async("my-baseline"))
|
|
151
|
+
diff = asyncio.run(sdk.compare_snapshot_async("my-baseline", baseline, ctx))
|
|
152
|
+
if diff.has_changes:
|
|
153
|
+
print(diff.summary()) # every change, with a line diff of changed inputs
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
## Test SDK
|
|
157
|
+
|
|
158
|
+
`VevalTestSdk` is a drop-in replacement that blocks real LLM calls, making it safe to use in unit tests. It still reports to the dashboard.
|
|
159
|
+
|
|
160
|
+
```python
|
|
161
|
+
from veval import VevalTestSdk
|
|
162
|
+
|
|
163
|
+
sdk = VevalTestSdk(options).with_replay(trace)
|
|
164
|
+
output = asyncio.run(sdk.run_async("my-agent", my_agent, input="test input"))
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
## API reference
|
|
168
|
+
|
|
169
|
+
### `VevalOptions`
|
|
170
|
+
|
|
171
|
+
| Parameter | Type | Default | Description |
|
|
172
|
+
|---|---|---|---|
|
|
173
|
+
| `api_key` | `str` | `""` | Your Veval API key — it also determines the workspace |
|
|
174
|
+
| `project_id` | `str` | `""` | Deprecated and ignored; will be removed |
|
|
175
|
+
| `flush_interval_ms` | `int` | `5000` | Batch flush interval |
|
|
176
|
+
| `flush_batch_size` | `int` | `50` | Max traces per flush |
|
|
177
|
+
|
|
178
|
+
### `VevalSdk`
|
|
179
|
+
|
|
180
|
+
| Method | Description |
|
|
181
|
+
|---|---|
|
|
182
|
+
| `run_async(name, callback, input)` | Run agent and send trace |
|
|
183
|
+
| `get_trace_async(trace_id)` | Fetch a recorded trace |
|
|
184
|
+
| `save_snapshot_async(name, trace_id_or_ctx)` | Store a named baseline (pins the trace) |
|
|
185
|
+
| `get_snapshot_async(name)` | Load the latest stored baseline |
|
|
186
|
+
| `load_snapshot_async(trace_id)` | Build a snapshot from a trace (not pinned) |
|
|
187
|
+
| `compare_snapshot_async(name, snapshot, ctx, options)` | Diff a run against a baseline and record it |
|
|
188
|
+
| `replay_async(trace, callback, options)` | Replay trace with mocked outputs |
|
|
189
|
+
| `run_scenario_async(name, agent, assertions, items)` | Run a test scenario |
|
|
190
|
+
|
|
191
|
+
### `TraceAssert` built-ins
|
|
192
|
+
|
|
193
|
+
| Assertion | Description |
|
|
194
|
+
|---|---|
|
|
195
|
+
| `no_errors()` | All steps must succeed |
|
|
196
|
+
| `max_steps(n)` | Total step count must not exceed `n` |
|
|
197
|
+
| `step_exists(name)` | A step with this name must appear |
|
|
198
|
+
| `max_cost(usd)` | Total cost must not exceed `usd` |
|
|
199
|
+
| `max_duration(ms)` | Total step duration must not exceed `ms` |
|
|
200
|
+
| `output_contains(text)` | At least one step output must contain `text` |
|
|
201
|
+
| `tool_called(name)` | A tool step with this name must appear |
|
|
202
|
+
|
|
203
|
+
Custom assertions implement `async ITraceAssertion.evaluate_async(ctx) -> Optional[str]` — return `None` to pass or an error string to fail.
|
|
204
|
+
|
|
205
|
+
## Requirements
|
|
206
|
+
|
|
207
|
+
- Python 3.10+
|
|
208
|
+
- No third-party dependencies
|
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
# veval-sdk
|
|
2
|
+
|
|
3
|
+
Python SDK for [Veval](https://veval.dev) — trace, evaluate, and test AI agents.
|
|
4
|
+
|
|
5
|
+
## Install
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
pip install veval-sdk
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
## Quick start
|
|
12
|
+
|
|
13
|
+
Wrap your agent with `run_async` to send a trace to the Veval dashboard:
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
import asyncio
|
|
17
|
+
from veval import VevalSdk, VevalOptions, VevalExecutionContext
|
|
18
|
+
|
|
19
|
+
sdk = VevalSdk(VevalOptions(api_key="veval-..."))
|
|
20
|
+
|
|
21
|
+
async def my_agent(ctx: VevalExecutionContext) -> str:
|
|
22
|
+
result = await ctx.track_step_async(
|
|
23
|
+
name="call-llm",
|
|
24
|
+
input=ctx.input,
|
|
25
|
+
step=lambda: call_my_llm(ctx.input),
|
|
26
|
+
)
|
|
27
|
+
return result
|
|
28
|
+
|
|
29
|
+
output = asyncio.run(sdk.run_async("my-agent", my_agent, input="Hello"))
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
Every call to `track_step_async` records the step name, input, output, timing, and any metadata you attach.
|
|
33
|
+
|
|
34
|
+
## Tracking steps
|
|
35
|
+
|
|
36
|
+
Use the `StepHandle` to attach token counts, cost, and model name:
|
|
37
|
+
|
|
38
|
+
```python
|
|
39
|
+
async def agent(ctx: VevalExecutionContext) -> str:
|
|
40
|
+
async def llm_call(handle):
|
|
41
|
+
response = await call_my_llm(ctx.input)
|
|
42
|
+
handle.set_meta("tokens_in", response.usage.input_tokens)
|
|
43
|
+
handle.set_meta("tokens_out", response.usage.output_tokens)
|
|
44
|
+
handle.set_meta("cost_usd", response.usage.cost)
|
|
45
|
+
handle.set_meta("model", "claude-opus-4-6")
|
|
46
|
+
return response.text
|
|
47
|
+
|
|
48
|
+
return await ctx.track_step_async("call-llm", ctx.input, llm_call)
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
## Assertions
|
|
52
|
+
|
|
53
|
+
Use `TraceAssert` to validate behaviour at the trace level:
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
from veval import TraceAssert
|
|
57
|
+
|
|
58
|
+
assertions = [
|
|
59
|
+
TraceAssert.no_errors(),
|
|
60
|
+
TraceAssert.max_steps(10),
|
|
61
|
+
TraceAssert.step_exists("call-llm"),
|
|
62
|
+
TraceAssert.max_cost(0.05),
|
|
63
|
+
TraceAssert.max_duration(30_000),
|
|
64
|
+
TraceAssert.output_contains("success"),
|
|
65
|
+
TraceAssert.tool_called("web-search"),
|
|
66
|
+
]
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
## Scenarios
|
|
70
|
+
|
|
71
|
+
Run a named scenario against a list of test items and assertions:
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
from veval import ScenarioItem
|
|
75
|
+
|
|
76
|
+
items = [
|
|
77
|
+
ScenarioItem(name="basic greeting", input="Hello"),
|
|
78
|
+
ScenarioItem(name="edge case", input=""),
|
|
79
|
+
]
|
|
80
|
+
|
|
81
|
+
result = asyncio.run(
|
|
82
|
+
sdk.run_scenario_async("smoke-test", my_agent, assertions, items=items)
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
print(f"Passed: {result.pass_count}/{len(result.results)}")
|
|
86
|
+
for r in result.results:
|
|
87
|
+
if not r.passed:
|
|
88
|
+
print(f" FAIL {r.item.name}: {r.failures}")
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
You can also pull items from the Veval dashboard by omitting `items`:
|
|
92
|
+
|
|
93
|
+
```python
|
|
94
|
+
result = asyncio.run(sdk.run_scenario_async("smoke-test", my_agent, assertions))
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
## Replay and snapshot testing
|
|
98
|
+
|
|
99
|
+
Load a previously recorded trace and replay it with mocked LLM responses — no real calls, deterministic results:
|
|
100
|
+
|
|
101
|
+
```python
|
|
102
|
+
trace = asyncio.run(sdk.get_trace_async("tr_abc123"))
|
|
103
|
+
|
|
104
|
+
replay = asyncio.run(sdk.replay_async(
|
|
105
|
+
trace,
|
|
106
|
+
my_agent,
|
|
107
|
+
ReplayOptions(mock_llm_responses=True, assertions=assertions),
|
|
108
|
+
))
|
|
109
|
+
|
|
110
|
+
print("Passed" if replay.passed else replay.failures)
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
Store a known-good run — every step with its input and output — and detect when a new run drifts from it: a changed prompt or tool argument, a step added, dropped, repeated, or reordered:
|
|
114
|
+
|
|
115
|
+
```python
|
|
116
|
+
# Save a baseline from a recorded trace. The trace is pinned, so retention never deletes it.
|
|
117
|
+
asyncio.run(sdk.save_snapshot_async("my-baseline", "tr_abc123"))
|
|
118
|
+
|
|
119
|
+
# In tests: an assertion like any other. A missing baseline fails; it never passes silently.
|
|
120
|
+
replay = asyncio.run(sdk.replay_async(trace, my_agent, ReplayOptions(
|
|
121
|
+
mock_llm_responses=True,
|
|
122
|
+
assertions=[TraceAssert.matches_snapshot(sdk, "my-baseline")],
|
|
123
|
+
compare_with_recording=SnapshotOptions(), # also fail if steps/inputs drift from the trace
|
|
124
|
+
)))
|
|
125
|
+
|
|
126
|
+
# Or compare directly, and record the result in the dashboard.
|
|
127
|
+
baseline = asyncio.run(sdk.get_snapshot_async("my-baseline"))
|
|
128
|
+
diff = asyncio.run(sdk.compare_snapshot_async("my-baseline", baseline, ctx))
|
|
129
|
+
if diff.has_changes:
|
|
130
|
+
print(diff.summary()) # every change, with a line diff of changed inputs
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
## Test SDK
|
|
134
|
+
|
|
135
|
+
`VevalTestSdk` is a drop-in replacement that blocks real LLM calls, making it safe to use in unit tests. It still reports to the dashboard.
|
|
136
|
+
|
|
137
|
+
```python
|
|
138
|
+
from veval import VevalTestSdk
|
|
139
|
+
|
|
140
|
+
sdk = VevalTestSdk(options).with_replay(trace)
|
|
141
|
+
output = asyncio.run(sdk.run_async("my-agent", my_agent, input="test input"))
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
## API reference
|
|
145
|
+
|
|
146
|
+
### `VevalOptions`
|
|
147
|
+
|
|
148
|
+
| Parameter | Type | Default | Description |
|
|
149
|
+
|---|---|---|---|
|
|
150
|
+
| `api_key` | `str` | `""` | Your Veval API key — it also determines the workspace |
|
|
151
|
+
| `project_id` | `str` | `""` | Deprecated and ignored; will be removed |
|
|
152
|
+
| `flush_interval_ms` | `int` | `5000` | Batch flush interval |
|
|
153
|
+
| `flush_batch_size` | `int` | `50` | Max traces per flush |
|
|
154
|
+
|
|
155
|
+
### `VevalSdk`
|
|
156
|
+
|
|
157
|
+
| Method | Description |
|
|
158
|
+
|---|---|
|
|
159
|
+
| `run_async(name, callback, input)` | Run agent and send trace |
|
|
160
|
+
| `get_trace_async(trace_id)` | Fetch a recorded trace |
|
|
161
|
+
| `save_snapshot_async(name, trace_id_or_ctx)` | Store a named baseline (pins the trace) |
|
|
162
|
+
| `get_snapshot_async(name)` | Load the latest stored baseline |
|
|
163
|
+
| `load_snapshot_async(trace_id)` | Build a snapshot from a trace (not pinned) |
|
|
164
|
+
| `compare_snapshot_async(name, snapshot, ctx, options)` | Diff a run against a baseline and record it |
|
|
165
|
+
| `replay_async(trace, callback, options)` | Replay trace with mocked outputs |
|
|
166
|
+
| `run_scenario_async(name, agent, assertions, items)` | Run a test scenario |
|
|
167
|
+
|
|
168
|
+
### `TraceAssert` built-ins
|
|
169
|
+
|
|
170
|
+
| Assertion | Description |
|
|
171
|
+
|---|---|
|
|
172
|
+
| `no_errors()` | All steps must succeed |
|
|
173
|
+
| `max_steps(n)` | Total step count must not exceed `n` |
|
|
174
|
+
| `step_exists(name)` | A step with this name must appear |
|
|
175
|
+
| `max_cost(usd)` | Total cost must not exceed `usd` |
|
|
176
|
+
| `max_duration(ms)` | Total step duration must not exceed `ms` |
|
|
177
|
+
| `output_contains(text)` | At least one step output must contain `text` |
|
|
178
|
+
| `tool_called(name)` | A tool step with this name must appear |
|
|
179
|
+
|
|
180
|
+
Custom assertions implement `async ITraceAssertion.evaluate_async(ctx) -> Optional[str]` — return `None` to pass or an error string to fail.
|
|
181
|
+
|
|
182
|
+
## Requirements
|
|
183
|
+
|
|
184
|
+
- Python 3.10+
|
|
185
|
+
- No third-party dependencies
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "veval-sdk"
|
|
7
|
+
version = "1.1.0"
|
|
8
|
+
description = "Trace, evaluate, and test AI agents with a few lines of code."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
keywords = ["ai", "agents", "llm", "evaluation", "testing", "tracing"]
|
|
13
|
+
classifiers = [
|
|
14
|
+
"Development Status :: 5 - Production/Stable",
|
|
15
|
+
"Intended Audience :: Developers",
|
|
16
|
+
"License :: OSI Approved :: MIT License",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Programming Language :: Python :: 3.10",
|
|
19
|
+
"Programming Language :: Python :: 3.11",
|
|
20
|
+
"Programming Language :: Python :: 3.12",
|
|
21
|
+
"Programming Language :: Python :: 3.13",
|
|
22
|
+
"Topic :: Software Development :: Testing",
|
|
23
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
24
|
+
]
|
|
25
|
+
dependencies = []
|
|
26
|
+
|
|
27
|
+
[tool.hatch.build.targets.wheel]
|
|
28
|
+
packages = ["veval"]
|
|
29
|
+
|
|
30
|
+
[project.urls]
|
|
31
|
+
Homepage = "https://veval.dev"
|
|
32
|
+
Documentation = "https://docs.veval.dev"
|
|
33
|
+
Repository = "https://github.com/JonathanDinelle/veval-sdks"
|
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
"""Core SDK tests. Run with `pytest sdk/python/tests` or `python sdk/python/tests/test_sdk.py`."""
|
|
2
|
+
import asyncio
|
|
3
|
+
import os
|
|
4
|
+
import sys
|
|
5
|
+
import warnings
|
|
6
|
+
from datetime import datetime, timezone
|
|
7
|
+
|
|
8
|
+
# Point every SDK at a closed local port: reporting calls fail fast and are swallowed, and nothing
|
|
9
|
+
# reaches the real API. Set before any SDK is constructed (the endpoint is read at construction).
|
|
10
|
+
os.environ["VEVAL_INTERNAL_ENDPOINT"] = "http://127.0.0.1:1"
|
|
11
|
+
sys.path.insert(0, os.path.join(os.path.dirname(__file__), ".."))
|
|
12
|
+
|
|
13
|
+
from veval import ( # noqa: E402
|
|
14
|
+
ReplayOptions, ScenarioItem, StepData, TraceAssert, TraceData, VevalExecutionContext,
|
|
15
|
+
VevalOptions, VevalSdk, VevalTestSdk,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def sdk():
|
|
20
|
+
return VevalSdk(VevalOptions(api_key="test"))
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def make_test_sdk():
|
|
24
|
+
return VevalTestSdk(VevalOptions(api_key="test"))
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def returning(value):
|
|
28
|
+
# A zero-argument callback: track_step_async passes a StepHandle to callbacks that take parameters.
|
|
29
|
+
async def produce():
|
|
30
|
+
return value
|
|
31
|
+
return produce
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def run(*steps):
|
|
35
|
+
async def go():
|
|
36
|
+
ctx = VevalExecutionContext("tr_run")
|
|
37
|
+
for name, inp, out in steps:
|
|
38
|
+
await ctx.track_step_async(name, inp, returning(out))
|
|
39
|
+
return ctx
|
|
40
|
+
return asyncio.run(go())
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def recording(*steps):
|
|
44
|
+
return TraceData(trace_id="tr_recorded", input="hello", status="success", steps=list(steps))
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def recorded(name, inp, out, type="llm"):
|
|
48
|
+
return StepData(step_id=name, name=name, type=type, input=inp, output=out, status="success")
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
# --- options -----------------------------------------------------------------------------------------
|
|
52
|
+
|
|
53
|
+
def test_endpoint_comes_from_env_not_a_public_option():
|
|
54
|
+
options = VevalOptions(api_key="k")
|
|
55
|
+
assert options._endpoint == "http://127.0.0.1:1"
|
|
56
|
+
assert not hasattr(options, "endpoint")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def test_project_id_is_deprecated_and_not_sent():
|
|
60
|
+
with warnings.catch_warnings(record=True) as caught:
|
|
61
|
+
warnings.simplefilter("always")
|
|
62
|
+
VevalOptions(api_key="k", project_id="p")
|
|
63
|
+
assert any(issubclass(w.category, DeprecationWarning) for w in caught)
|
|
64
|
+
|
|
65
|
+
now = datetime.now(timezone.utc)
|
|
66
|
+
payload = sdk()._build_payload("tr", "agent", run(("s", "in", "out")), None, None, "success", None, now, now)
|
|
67
|
+
assert "project_id" not in payload
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
# --- running and steps -------------------------------------------------------------------------------
|
|
71
|
+
|
|
72
|
+
def test_run_async_returns_result_and_records_steps_with_metadata():
|
|
73
|
+
seen = {}
|
|
74
|
+
|
|
75
|
+
async def agent(ctx):
|
|
76
|
+
seen["ctx"] = ctx
|
|
77
|
+
ctx.set_metadata("user_id", "u_1")
|
|
78
|
+
|
|
79
|
+
async def call(step):
|
|
80
|
+
step.set_meta("type", "llm")
|
|
81
|
+
step.set_meta("cost_usd", 0.01)
|
|
82
|
+
step.set_meta("provider", "anthropic")
|
|
83
|
+
return "answer"
|
|
84
|
+
return await ctx.track_step_async("call-llm", "prompt", call)
|
|
85
|
+
|
|
86
|
+
assert asyncio.run(sdk().run_async("agent", agent)) == "answer"
|
|
87
|
+
step = seen["ctx"].steps[0]
|
|
88
|
+
assert (step.name, step.type, step.cost_usd, step.output, step.status) == ("call-llm", "llm", 0.01, "answer", "success")
|
|
89
|
+
assert step.metadata["provider"] == "anthropic"
|
|
90
|
+
assert seen["ctx"].trace_meta["user_id"] == "u_1"
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def test_run_async_reraises_agent_errors_and_marks_step_failed():
|
|
94
|
+
seen = {}
|
|
95
|
+
|
|
96
|
+
async def boom():
|
|
97
|
+
raise RuntimeError("kaboom")
|
|
98
|
+
|
|
99
|
+
async def agent(ctx):
|
|
100
|
+
seen["ctx"] = ctx
|
|
101
|
+
return await ctx.track_step_async("boom", None, boom)
|
|
102
|
+
|
|
103
|
+
try:
|
|
104
|
+
asyncio.run(sdk().run_async("agent", agent))
|
|
105
|
+
raise AssertionError("expected the agent error to propagate")
|
|
106
|
+
except RuntimeError as ex:
|
|
107
|
+
assert str(ex) == "kaboom"
|
|
108
|
+
assert (seen["ctx"].steps[0].status, seen["ctx"].steps[0].error) == ("error", "kaboom")
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
# --- replay ------------------------------------------------------------------------------------------
|
|
112
|
+
|
|
113
|
+
def test_replay_serves_recorded_outputs_and_never_runs_the_step():
|
|
114
|
+
ran = []
|
|
115
|
+
|
|
116
|
+
async def live():
|
|
117
|
+
ran.append(True)
|
|
118
|
+
return "live"
|
|
119
|
+
|
|
120
|
+
async def agent(ctx):
|
|
121
|
+
return await ctx.track_step_async("classify", "hello", live)
|
|
122
|
+
|
|
123
|
+
result = asyncio.run(sdk().replay_async(recording(recorded("classify", "hello", "greeting")), agent, ReplayOptions(mock_llm_responses=True)))
|
|
124
|
+
assert result.output == "greeting"
|
|
125
|
+
assert ran == []
|
|
126
|
+
assert result.passed
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def test_replay_fails_when_a_step_has_no_recording_instead_of_calling_live():
|
|
130
|
+
async def agent(ctx):
|
|
131
|
+
return await ctx.track_step_async("unrecorded", None, returning("live"))
|
|
132
|
+
|
|
133
|
+
result = asyncio.run(sdk().replay_async(recording(recorded("classify", "hello", "greeting")), agent, ReplayOptions(mock_llm_responses=True)))
|
|
134
|
+
assert not result.passed
|
|
135
|
+
assert "no mock output for step 'unrecorded'" in result.failures[0]
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def test_with_replay_makes_the_production_entry_point_replay_unchanged():
|
|
139
|
+
replaying = make_test_sdk().with_replay(recording(recorded("classify", "hello", "greeting")))
|
|
140
|
+
|
|
141
|
+
async def agent(ctx):
|
|
142
|
+
return await ctx.track_step_async("classify", "hello", returning("live"))
|
|
143
|
+
|
|
144
|
+
assert asyncio.run(replaying.run_async("agent", agent)) == "greeting"
|
|
145
|
+
assert replaying.last_status == "success"
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
# --- assertions --------------------------------------------------------------------------------------
|
|
149
|
+
|
|
150
|
+
def test_built_in_assertions_pass_and_fail_as_documented():
|
|
151
|
+
async def build():
|
|
152
|
+
ctx = VevalExecutionContext("tr")
|
|
153
|
+
|
|
154
|
+
async def fetch(step):
|
|
155
|
+
step.set_meta("type", "tool")
|
|
156
|
+
step.set_meta("cost_usd", 0.2)
|
|
157
|
+
return "Montreal weather"
|
|
158
|
+
await ctx.track_step_async("fetch", "q", fetch)
|
|
159
|
+
await ctx.track_step_async("write", "d", returning("briefing"))
|
|
160
|
+
return ctx
|
|
161
|
+
ctx = asyncio.run(build())
|
|
162
|
+
check = lambda a: asyncio.run(a.evaluate_async(ctx)) # noqa: E731
|
|
163
|
+
|
|
164
|
+
assert check(TraceAssert.no_errors()) is None
|
|
165
|
+
assert check(TraceAssert.max_steps(2)) is None
|
|
166
|
+
assert check(TraceAssert.max_steps(1)).startswith("MaxSteps")
|
|
167
|
+
assert check(TraceAssert.step_exists("write")) is None
|
|
168
|
+
assert check(TraceAssert.step_exists("nope")).startswith("StepExists")
|
|
169
|
+
assert check(TraceAssert.tool_called("fetch")) is None
|
|
170
|
+
assert check(TraceAssert.tool_called("write")).startswith("ToolCalled")
|
|
171
|
+
assert check(TraceAssert.max_cost(0.1)).startswith("MaxCost")
|
|
172
|
+
assert check(TraceAssert.output_contains("Montreal")) is None
|
|
173
|
+
assert check(TraceAssert.output_contains("Paris")).startswith("OutputContains")
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
# --- judge -------------------------------------------------------------------------------------------
|
|
177
|
+
|
|
178
|
+
def test_judge_mock_passes_records_verdict_and_reports_failures():
|
|
179
|
+
judged = make_test_sdk().with_judge_mock("polite", True, 0.9, "Friendly.").with_judge_mock("concise", False, 0.4, "Rambles.")
|
|
180
|
+
ctx = run(("reply", "hi", "Hello there!"))
|
|
181
|
+
|
|
182
|
+
assert asyncio.run(TraceAssert.judge(judged, "polite").evaluate_async(ctx)) is None
|
|
183
|
+
assert asyncio.run(TraceAssert.judge(judged, "concise").evaluate_async(ctx)) == "Judge: concise (score 0.40) — Rambles."
|
|
184
|
+
assert [(j["criteria"], j["passed"]) for j in ctx.judgments] == [("polite", True), ("concise", False)]
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def test_judge_without_a_mock_fails_instead_of_making_a_billed_call():
|
|
188
|
+
failure = asyncio.run(TraceAssert.judge(make_test_sdk(), "unmocked").evaluate_async(run(("reply", "hi", "yo"))))
|
|
189
|
+
assert failure.startswith("Judge: evaluation failed")
|
|
190
|
+
assert "no judge mock for criteria 'unmocked'" in failure
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
# --- scenarios ---------------------------------------------------------------------------------------
|
|
194
|
+
|
|
195
|
+
def test_scenario_runs_each_item_with_shared_and_per_item_assertions():
|
|
196
|
+
async def agent(ctx):
|
|
197
|
+
async def reply():
|
|
198
|
+
if ctx.input == "crash":
|
|
199
|
+
raise RuntimeError("agent crashed")
|
|
200
|
+
return f"echo: {ctx.input}"
|
|
201
|
+
return await ctx.track_step_async("reply", ctx.input, reply)
|
|
202
|
+
|
|
203
|
+
result = asyncio.run(make_test_sdk().run_scenario_async("echo", agent, [TraceAssert.step_exists("reply")], [
|
|
204
|
+
ScenarioItem(name="ok", input="hi", assertions=[TraceAssert.output_contains("echo: hi")]),
|
|
205
|
+
ScenarioItem(name="wrong output", input="bye", assertions=[TraceAssert.output_contains("echo: hi")]),
|
|
206
|
+
ScenarioItem(name="crash", input="crash"),
|
|
207
|
+
]))
|
|
208
|
+
|
|
209
|
+
assert [r.passed for r in result.results] == [True, False, False]
|
|
210
|
+
assert result.pass_count == 1
|
|
211
|
+
assert any("agent crashed" in f for f in result.results[2].failures)
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
if __name__ == "__main__":
|
|
215
|
+
tests = [f for name, f in sorted(globals().items()) if name.startswith("test_")]
|
|
216
|
+
for t in tests:
|
|
217
|
+
t()
|
|
218
|
+
print(f"python sdk tests: {len(tests)} passed")
|