veval-sdk 1.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,9 @@
1
+ bin/
2
+ obj/
3
+ node_modules/
4
+ dist/
5
+ __pycache__/
6
+ *.egg-info/
7
+ .venv/
8
+ build/
9
+ *.pyc
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Jonathan Dinelle
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,208 @@
1
+ Metadata-Version: 2.5
2
+ Name: veval-sdk
3
+ Version: 1.1.0
4
+ Summary: Trace, evaluate, and test AI agents with a few lines of code.
5
+ Project-URL: Homepage, https://veval.dev
6
+ Project-URL: Documentation, https://docs.veval.dev
7
+ Project-URL: Repository, https://github.com/JonathanDinelle/veval-sdks
8
+ License: MIT
9
+ License-File: LICENSE
10
+ Keywords: agents,ai,evaluation,llm,testing,tracing
11
+ Classifier: Development Status :: 5 - Production/Stable
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.10
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Programming Language :: Python :: 3.13
19
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
20
+ Classifier: Topic :: Software Development :: Testing
21
+ Requires-Python: >=3.10
22
+ Description-Content-Type: text/markdown
23
+
24
+ # veval-sdk
25
+
26
+ Python SDK for [Veval](https://veval.dev) — trace, evaluate, and test AI agents.
27
+
28
+ ## Install
29
+
30
+ ```bash
31
+ pip install veval-sdk
32
+ ```
33
+
34
+ ## Quick start
35
+
36
+ Wrap your agent with `run_async` to send a trace to the Veval dashboard:
37
+
38
+ ```python
39
+ import asyncio
40
+ from veval import VevalSdk, VevalOptions, VevalExecutionContext
41
+
42
+ sdk = VevalSdk(VevalOptions(api_key="veval-..."))
43
+
44
+ async def my_agent(ctx: VevalExecutionContext) -> str:
45
+ result = await ctx.track_step_async(
46
+ name="call-llm",
47
+ input=ctx.input,
48
+ step=lambda: call_my_llm(ctx.input),
49
+ )
50
+ return result
51
+
52
+ output = asyncio.run(sdk.run_async("my-agent", my_agent, input="Hello"))
53
+ ```
54
+
55
+ Every call to `track_step_async` records the step name, input, output, timing, and any metadata you attach.
56
+
57
+ ## Tracking steps
58
+
59
+ Use the `StepHandle` to attach token counts, cost, and model name:
60
+
61
+ ```python
62
+ async def agent(ctx: VevalExecutionContext) -> str:
63
+ async def llm_call(handle):
64
+ response = await call_my_llm(ctx.input)
65
+ handle.set_meta("tokens_in", response.usage.input_tokens)
66
+ handle.set_meta("tokens_out", response.usage.output_tokens)
67
+ handle.set_meta("cost_usd", response.usage.cost)
68
+ handle.set_meta("model", "claude-opus-4-6")
69
+ return response.text
70
+
71
+ return await ctx.track_step_async("call-llm", ctx.input, llm_call)
72
+ ```
73
+
74
+ ## Assertions
75
+
76
+ Use `TraceAssert` to validate behaviour at the trace level:
77
+
78
+ ```python
79
+ from veval import TraceAssert
80
+
81
+ assertions = [
82
+ TraceAssert.no_errors(),
83
+ TraceAssert.max_steps(10),
84
+ TraceAssert.step_exists("call-llm"),
85
+ TraceAssert.max_cost(0.05),
86
+ TraceAssert.max_duration(30_000),
87
+ TraceAssert.output_contains("success"),
88
+ TraceAssert.tool_called("web-search"),
89
+ ]
90
+ ```
91
+
92
+ ## Scenarios
93
+
94
+ Run a named scenario against a list of test items and assertions:
95
+
96
+ ```python
97
+ from veval import ScenarioItem
98
+
99
+ items = [
100
+ ScenarioItem(name="basic greeting", input="Hello"),
101
+ ScenarioItem(name="edge case", input=""),
102
+ ]
103
+
104
+ result = asyncio.run(
105
+ sdk.run_scenario_async("smoke-test", my_agent, assertions, items=items)
106
+ )
107
+
108
+ print(f"Passed: {result.pass_count}/{len(result.results)}")
109
+ for r in result.results:
110
+ if not r.passed:
111
+ print(f" FAIL {r.item.name}: {r.failures}")
112
+ ```
113
+
114
+ You can also pull items from the Veval dashboard by omitting `items`:
115
+
116
+ ```python
117
+ result = asyncio.run(sdk.run_scenario_async("smoke-test", my_agent, assertions))
118
+ ```
119
+
120
+ ## Replay and snapshot testing
121
+
122
+ Load a previously recorded trace and replay it with mocked LLM responses — no real calls, deterministic results:
123
+
124
+ ```python
125
+ trace = asyncio.run(sdk.get_trace_async("tr_abc123"))
126
+
127
+ replay = asyncio.run(sdk.replay_async(
128
+ trace,
129
+ my_agent,
130
+ ReplayOptions(mock_llm_responses=True, assertions=assertions),
131
+ ))
132
+
133
+ print("Passed" if replay.passed else replay.failures)
134
+ ```
135
+
136
+ Store a known-good run — every step with its input and output — and detect when a new run drifts from it: a changed prompt or tool argument, a step added, dropped, repeated, or reordered:
137
+
138
+ ```python
139
+ # Save a baseline from a recorded trace. The trace is pinned, so retention never deletes it.
140
+ asyncio.run(sdk.save_snapshot_async("my-baseline", "tr_abc123"))
141
+
142
+ # In tests: an assertion like any other. A missing baseline fails; it never passes silently.
143
+ replay = asyncio.run(sdk.replay_async(trace, my_agent, ReplayOptions(
144
+ mock_llm_responses=True,
145
+ assertions=[TraceAssert.matches_snapshot(sdk, "my-baseline")],
146
+ compare_with_recording=SnapshotOptions(), # also fail if steps/inputs drift from the trace
147
+ )))
148
+
149
+ # Or compare directly, and record the result in the dashboard.
150
+ baseline = asyncio.run(sdk.get_snapshot_async("my-baseline"))
151
+ diff = asyncio.run(sdk.compare_snapshot_async("my-baseline", baseline, ctx))
152
+ if diff.has_changes:
153
+ print(diff.summary()) # every change, with a line diff of changed inputs
154
+ ```
155
+
156
+ ## Test SDK
157
+
158
+ `VevalTestSdk` is a drop-in replacement that blocks real LLM calls, making it safe to use in unit tests. It still reports to the dashboard.
159
+
160
+ ```python
161
+ from veval import VevalTestSdk
162
+
163
+ sdk = VevalTestSdk(options).with_replay(trace)
164
+ output = asyncio.run(sdk.run_async("my-agent", my_agent, input="test input"))
165
+ ```
166
+
167
+ ## API reference
168
+
169
+ ### `VevalOptions`
170
+
171
+ | Parameter | Type | Default | Description |
172
+ |---|---|---|---|
173
+ | `api_key` | `str` | `""` | Your Veval API key — it also determines the workspace |
174
+ | `project_id` | `str` | `""` | Deprecated and ignored; will be removed |
175
+ | `flush_interval_ms` | `int` | `5000` | Batch flush interval |
176
+ | `flush_batch_size` | `int` | `50` | Max traces per flush |
177
+
178
+ ### `VevalSdk`
179
+
180
+ | Method | Description |
181
+ |---|---|
182
+ | `run_async(name, callback, input)` | Run agent and send trace |
183
+ | `get_trace_async(trace_id)` | Fetch a recorded trace |
184
+ | `save_snapshot_async(name, trace_id_or_ctx)` | Store a named baseline (pins the trace) |
185
+ | `get_snapshot_async(name)` | Load the latest stored baseline |
186
+ | `load_snapshot_async(trace_id)` | Build a snapshot from a trace (not pinned) |
187
+ | `compare_snapshot_async(name, snapshot, ctx, options)` | Diff a run against a baseline and record it |
188
+ | `replay_async(trace, callback, options)` | Replay trace with mocked outputs |
189
+ | `run_scenario_async(name, agent, assertions, items)` | Run a test scenario |
190
+
191
+ ### `TraceAssert` built-ins
192
+
193
+ | Assertion | Description |
194
+ |---|---|
195
+ | `no_errors()` | All steps must succeed |
196
+ | `max_steps(n)` | Total step count must not exceed `n` |
197
+ | `step_exists(name)` | A step with this name must appear |
198
+ | `max_cost(usd)` | Total cost must not exceed `usd` |
199
+ | `max_duration(ms)` | Total step duration must not exceed `ms` |
200
+ | `output_contains(text)` | At least one step output must contain `text` |
201
+ | `tool_called(name)` | A tool step with this name must appear |
202
+
203
+ Custom assertions implement `async ITraceAssertion.evaluate_async(ctx) -> Optional[str]` — return `None` to pass or an error string to fail.
204
+
205
+ ## Requirements
206
+
207
+ - Python 3.10+
208
+ - No third-party dependencies
@@ -0,0 +1,185 @@
1
+ # veval-sdk
2
+
3
+ Python SDK for [Veval](https://veval.dev) — trace, evaluate, and test AI agents.
4
+
5
+ ## Install
6
+
7
+ ```bash
8
+ pip install veval-sdk
9
+ ```
10
+
11
+ ## Quick start
12
+
13
+ Wrap your agent with `run_async` to send a trace to the Veval dashboard:
14
+
15
+ ```python
16
+ import asyncio
17
+ from veval import VevalSdk, VevalOptions, VevalExecutionContext
18
+
19
+ sdk = VevalSdk(VevalOptions(api_key="veval-..."))
20
+
21
+ async def my_agent(ctx: VevalExecutionContext) -> str:
22
+ result = await ctx.track_step_async(
23
+ name="call-llm",
24
+ input=ctx.input,
25
+ step=lambda: call_my_llm(ctx.input),
26
+ )
27
+ return result
28
+
29
+ output = asyncio.run(sdk.run_async("my-agent", my_agent, input="Hello"))
30
+ ```
31
+
32
+ Every call to `track_step_async` records the step name, input, output, timing, and any metadata you attach.
33
+
34
+ ## Tracking steps
35
+
36
+ Use the `StepHandle` to attach token counts, cost, and model name:
37
+
38
+ ```python
39
+ async def agent(ctx: VevalExecutionContext) -> str:
40
+ async def llm_call(handle):
41
+ response = await call_my_llm(ctx.input)
42
+ handle.set_meta("tokens_in", response.usage.input_tokens)
43
+ handle.set_meta("tokens_out", response.usage.output_tokens)
44
+ handle.set_meta("cost_usd", response.usage.cost)
45
+ handle.set_meta("model", "claude-opus-4-6")
46
+ return response.text
47
+
48
+ return await ctx.track_step_async("call-llm", ctx.input, llm_call)
49
+ ```
50
+
51
+ ## Assertions
52
+
53
+ Use `TraceAssert` to validate behaviour at the trace level:
54
+
55
+ ```python
56
+ from veval import TraceAssert
57
+
58
+ assertions = [
59
+ TraceAssert.no_errors(),
60
+ TraceAssert.max_steps(10),
61
+ TraceAssert.step_exists("call-llm"),
62
+ TraceAssert.max_cost(0.05),
63
+ TraceAssert.max_duration(30_000),
64
+ TraceAssert.output_contains("success"),
65
+ TraceAssert.tool_called("web-search"),
66
+ ]
67
+ ```
68
+
69
+ ## Scenarios
70
+
71
+ Run a named scenario against a list of test items and assertions:
72
+
73
+ ```python
74
+ from veval import ScenarioItem
75
+
76
+ items = [
77
+ ScenarioItem(name="basic greeting", input="Hello"),
78
+ ScenarioItem(name="edge case", input=""),
79
+ ]
80
+
81
+ result = asyncio.run(
82
+ sdk.run_scenario_async("smoke-test", my_agent, assertions, items=items)
83
+ )
84
+
85
+ print(f"Passed: {result.pass_count}/{len(result.results)}")
86
+ for r in result.results:
87
+ if not r.passed:
88
+ print(f" FAIL {r.item.name}: {r.failures}")
89
+ ```
90
+
91
+ You can also pull items from the Veval dashboard by omitting `items`:
92
+
93
+ ```python
94
+ result = asyncio.run(sdk.run_scenario_async("smoke-test", my_agent, assertions))
95
+ ```
96
+
97
+ ## Replay and snapshot testing
98
+
99
+ Load a previously recorded trace and replay it with mocked LLM responses — no real calls, deterministic results:
100
+
101
+ ```python
102
+ trace = asyncio.run(sdk.get_trace_async("tr_abc123"))
103
+
104
+ replay = asyncio.run(sdk.replay_async(
105
+ trace,
106
+ my_agent,
107
+ ReplayOptions(mock_llm_responses=True, assertions=assertions),
108
+ ))
109
+
110
+ print("Passed" if replay.passed else replay.failures)
111
+ ```
112
+
113
+ Store a known-good run — every step with its input and output — and detect when a new run drifts from it: a changed prompt or tool argument, a step added, dropped, repeated, or reordered:
114
+
115
+ ```python
116
+ # Save a baseline from a recorded trace. The trace is pinned, so retention never deletes it.
117
+ asyncio.run(sdk.save_snapshot_async("my-baseline", "tr_abc123"))
118
+
119
+ # In tests: an assertion like any other. A missing baseline fails; it never passes silently.
120
+ replay = asyncio.run(sdk.replay_async(trace, my_agent, ReplayOptions(
121
+ mock_llm_responses=True,
122
+ assertions=[TraceAssert.matches_snapshot(sdk, "my-baseline")],
123
+ compare_with_recording=SnapshotOptions(), # also fail if steps/inputs drift from the trace
124
+ )))
125
+
126
+ # Or compare directly, and record the result in the dashboard.
127
+ baseline = asyncio.run(sdk.get_snapshot_async("my-baseline"))
128
+ diff = asyncio.run(sdk.compare_snapshot_async("my-baseline", baseline, ctx))
129
+ if diff.has_changes:
130
+ print(diff.summary()) # every change, with a line diff of changed inputs
131
+ ```
132
+
133
+ ## Test SDK
134
+
135
+ `VevalTestSdk` is a drop-in replacement that blocks real LLM calls, making it safe to use in unit tests. It still reports to the dashboard.
136
+
137
+ ```python
138
+ from veval import VevalTestSdk
139
+
140
+ sdk = VevalTestSdk(options).with_replay(trace)
141
+ output = asyncio.run(sdk.run_async("my-agent", my_agent, input="test input"))
142
+ ```
143
+
144
+ ## API reference
145
+
146
+ ### `VevalOptions`
147
+
148
+ | Parameter | Type | Default | Description |
149
+ |---|---|---|---|
150
+ | `api_key` | `str` | `""` | Your Veval API key — it also determines the workspace |
151
+ | `project_id` | `str` | `""` | Deprecated and ignored; will be removed |
152
+ | `flush_interval_ms` | `int` | `5000` | Batch flush interval |
153
+ | `flush_batch_size` | `int` | `50` | Max traces per flush |
154
+
155
+ ### `VevalSdk`
156
+
157
+ | Method | Description |
158
+ |---|---|
159
+ | `run_async(name, callback, input)` | Run agent and send trace |
160
+ | `get_trace_async(trace_id)` | Fetch a recorded trace |
161
+ | `save_snapshot_async(name, trace_id_or_ctx)` | Store a named baseline (pins the trace) |
162
+ | `get_snapshot_async(name)` | Load the latest stored baseline |
163
+ | `load_snapshot_async(trace_id)` | Build a snapshot from a trace (not pinned) |
164
+ | `compare_snapshot_async(name, snapshot, ctx, options)` | Diff a run against a baseline and record it |
165
+ | `replay_async(trace, callback, options)` | Replay trace with mocked outputs |
166
+ | `run_scenario_async(name, agent, assertions, items)` | Run a test scenario |
167
+
168
+ ### `TraceAssert` built-ins
169
+
170
+ | Assertion | Description |
171
+ |---|---|
172
+ | `no_errors()` | All steps must succeed |
173
+ | `max_steps(n)` | Total step count must not exceed `n` |
174
+ | `step_exists(name)` | A step with this name must appear |
175
+ | `max_cost(usd)` | Total cost must not exceed `usd` |
176
+ | `max_duration(ms)` | Total step duration must not exceed `ms` |
177
+ | `output_contains(text)` | At least one step output must contain `text` |
178
+ | `tool_called(name)` | A tool step with this name must appear |
179
+
180
+ Custom assertions implement `async ITraceAssertion.evaluate_async(ctx) -> Optional[str]` — return `None` to pass or an error string to fail.
181
+
182
+ ## Requirements
183
+
184
+ - Python 3.10+
185
+ - No third-party dependencies
@@ -0,0 +1,33 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "veval-sdk"
7
+ version = "1.1.0"
8
+ description = "Trace, evaluate, and test AI agents with a few lines of code."
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = { text = "MIT" }
12
+ keywords = ["ai", "agents", "llm", "evaluation", "testing", "tracing"]
13
+ classifiers = [
14
+ "Development Status :: 5 - Production/Stable",
15
+ "Intended Audience :: Developers",
16
+ "License :: OSI Approved :: MIT License",
17
+ "Programming Language :: Python :: 3",
18
+ "Programming Language :: Python :: 3.10",
19
+ "Programming Language :: Python :: 3.11",
20
+ "Programming Language :: Python :: 3.12",
21
+ "Programming Language :: Python :: 3.13",
22
+ "Topic :: Software Development :: Testing",
23
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
24
+ ]
25
+ dependencies = []
26
+
27
+ [tool.hatch.build.targets.wheel]
28
+ packages = ["veval"]
29
+
30
+ [project.urls]
31
+ Homepage = "https://veval.dev"
32
+ Documentation = "https://docs.veval.dev"
33
+ Repository = "https://github.com/JonathanDinelle/veval-sdks"
@@ -0,0 +1,218 @@
1
+ """Core SDK tests. Run with `pytest sdk/python/tests` or `python sdk/python/tests/test_sdk.py`."""
2
+ import asyncio
3
+ import os
4
+ import sys
5
+ import warnings
6
+ from datetime import datetime, timezone
7
+
8
+ # Point every SDK at a closed local port: reporting calls fail fast and are swallowed, and nothing
9
+ # reaches the real API. Set before any SDK is constructed (the endpoint is read at construction).
10
+ os.environ["VEVAL_INTERNAL_ENDPOINT"] = "http://127.0.0.1:1"
11
+ sys.path.insert(0, os.path.join(os.path.dirname(__file__), ".."))
12
+
13
+ from veval import ( # noqa: E402
14
+ ReplayOptions, ScenarioItem, StepData, TraceAssert, TraceData, VevalExecutionContext,
15
+ VevalOptions, VevalSdk, VevalTestSdk,
16
+ )
17
+
18
+
19
+ def sdk():
20
+ return VevalSdk(VevalOptions(api_key="test"))
21
+
22
+
23
+ def make_test_sdk():
24
+ return VevalTestSdk(VevalOptions(api_key="test"))
25
+
26
+
27
+ def returning(value):
28
+ # A zero-argument callback: track_step_async passes a StepHandle to callbacks that take parameters.
29
+ async def produce():
30
+ return value
31
+ return produce
32
+
33
+
34
+ def run(*steps):
35
+ async def go():
36
+ ctx = VevalExecutionContext("tr_run")
37
+ for name, inp, out in steps:
38
+ await ctx.track_step_async(name, inp, returning(out))
39
+ return ctx
40
+ return asyncio.run(go())
41
+
42
+
43
+ def recording(*steps):
44
+ return TraceData(trace_id="tr_recorded", input="hello", status="success", steps=list(steps))
45
+
46
+
47
+ def recorded(name, inp, out, type="llm"):
48
+ return StepData(step_id=name, name=name, type=type, input=inp, output=out, status="success")
49
+
50
+
51
+ # --- options -----------------------------------------------------------------------------------------
52
+
53
+ def test_endpoint_comes_from_env_not_a_public_option():
54
+ options = VevalOptions(api_key="k")
55
+ assert options._endpoint == "http://127.0.0.1:1"
56
+ assert not hasattr(options, "endpoint")
57
+
58
+
59
+ def test_project_id_is_deprecated_and_not_sent():
60
+ with warnings.catch_warnings(record=True) as caught:
61
+ warnings.simplefilter("always")
62
+ VevalOptions(api_key="k", project_id="p")
63
+ assert any(issubclass(w.category, DeprecationWarning) for w in caught)
64
+
65
+ now = datetime.now(timezone.utc)
66
+ payload = sdk()._build_payload("tr", "agent", run(("s", "in", "out")), None, None, "success", None, now, now)
67
+ assert "project_id" not in payload
68
+
69
+
70
+ # --- running and steps -------------------------------------------------------------------------------
71
+
72
+ def test_run_async_returns_result_and_records_steps_with_metadata():
73
+ seen = {}
74
+
75
+ async def agent(ctx):
76
+ seen["ctx"] = ctx
77
+ ctx.set_metadata("user_id", "u_1")
78
+
79
+ async def call(step):
80
+ step.set_meta("type", "llm")
81
+ step.set_meta("cost_usd", 0.01)
82
+ step.set_meta("provider", "anthropic")
83
+ return "answer"
84
+ return await ctx.track_step_async("call-llm", "prompt", call)
85
+
86
+ assert asyncio.run(sdk().run_async("agent", agent)) == "answer"
87
+ step = seen["ctx"].steps[0]
88
+ assert (step.name, step.type, step.cost_usd, step.output, step.status) == ("call-llm", "llm", 0.01, "answer", "success")
89
+ assert step.metadata["provider"] == "anthropic"
90
+ assert seen["ctx"].trace_meta["user_id"] == "u_1"
91
+
92
+
93
+ def test_run_async_reraises_agent_errors_and_marks_step_failed():
94
+ seen = {}
95
+
96
+ async def boom():
97
+ raise RuntimeError("kaboom")
98
+
99
+ async def agent(ctx):
100
+ seen["ctx"] = ctx
101
+ return await ctx.track_step_async("boom", None, boom)
102
+
103
+ try:
104
+ asyncio.run(sdk().run_async("agent", agent))
105
+ raise AssertionError("expected the agent error to propagate")
106
+ except RuntimeError as ex:
107
+ assert str(ex) == "kaboom"
108
+ assert (seen["ctx"].steps[0].status, seen["ctx"].steps[0].error) == ("error", "kaboom")
109
+
110
+
111
+ # --- replay ------------------------------------------------------------------------------------------
112
+
113
+ def test_replay_serves_recorded_outputs_and_never_runs_the_step():
114
+ ran = []
115
+
116
+ async def live():
117
+ ran.append(True)
118
+ return "live"
119
+
120
+ async def agent(ctx):
121
+ return await ctx.track_step_async("classify", "hello", live)
122
+
123
+ result = asyncio.run(sdk().replay_async(recording(recorded("classify", "hello", "greeting")), agent, ReplayOptions(mock_llm_responses=True)))
124
+ assert result.output == "greeting"
125
+ assert ran == []
126
+ assert result.passed
127
+
128
+
129
+ def test_replay_fails_when_a_step_has_no_recording_instead_of_calling_live():
130
+ async def agent(ctx):
131
+ return await ctx.track_step_async("unrecorded", None, returning("live"))
132
+
133
+ result = asyncio.run(sdk().replay_async(recording(recorded("classify", "hello", "greeting")), agent, ReplayOptions(mock_llm_responses=True)))
134
+ assert not result.passed
135
+ assert "no mock output for step 'unrecorded'" in result.failures[0]
136
+
137
+
138
+ def test_with_replay_makes_the_production_entry_point_replay_unchanged():
139
+ replaying = make_test_sdk().with_replay(recording(recorded("classify", "hello", "greeting")))
140
+
141
+ async def agent(ctx):
142
+ return await ctx.track_step_async("classify", "hello", returning("live"))
143
+
144
+ assert asyncio.run(replaying.run_async("agent", agent)) == "greeting"
145
+ assert replaying.last_status == "success"
146
+
147
+
148
+ # --- assertions --------------------------------------------------------------------------------------
149
+
150
+ def test_built_in_assertions_pass_and_fail_as_documented():
151
+ async def build():
152
+ ctx = VevalExecutionContext("tr")
153
+
154
+ async def fetch(step):
155
+ step.set_meta("type", "tool")
156
+ step.set_meta("cost_usd", 0.2)
157
+ return "Montreal weather"
158
+ await ctx.track_step_async("fetch", "q", fetch)
159
+ await ctx.track_step_async("write", "d", returning("briefing"))
160
+ return ctx
161
+ ctx = asyncio.run(build())
162
+ check = lambda a: asyncio.run(a.evaluate_async(ctx)) # noqa: E731
163
+
164
+ assert check(TraceAssert.no_errors()) is None
165
+ assert check(TraceAssert.max_steps(2)) is None
166
+ assert check(TraceAssert.max_steps(1)).startswith("MaxSteps")
167
+ assert check(TraceAssert.step_exists("write")) is None
168
+ assert check(TraceAssert.step_exists("nope")).startswith("StepExists")
169
+ assert check(TraceAssert.tool_called("fetch")) is None
170
+ assert check(TraceAssert.tool_called("write")).startswith("ToolCalled")
171
+ assert check(TraceAssert.max_cost(0.1)).startswith("MaxCost")
172
+ assert check(TraceAssert.output_contains("Montreal")) is None
173
+ assert check(TraceAssert.output_contains("Paris")).startswith("OutputContains")
174
+
175
+
176
+ # --- judge -------------------------------------------------------------------------------------------
177
+
178
+ def test_judge_mock_passes_records_verdict_and_reports_failures():
179
+ judged = make_test_sdk().with_judge_mock("polite", True, 0.9, "Friendly.").with_judge_mock("concise", False, 0.4, "Rambles.")
180
+ ctx = run(("reply", "hi", "Hello there!"))
181
+
182
+ assert asyncio.run(TraceAssert.judge(judged, "polite").evaluate_async(ctx)) is None
183
+ assert asyncio.run(TraceAssert.judge(judged, "concise").evaluate_async(ctx)) == "Judge: concise (score 0.40) — Rambles."
184
+ assert [(j["criteria"], j["passed"]) for j in ctx.judgments] == [("polite", True), ("concise", False)]
185
+
186
+
187
+ def test_judge_without_a_mock_fails_instead_of_making_a_billed_call():
188
+ failure = asyncio.run(TraceAssert.judge(make_test_sdk(), "unmocked").evaluate_async(run(("reply", "hi", "yo"))))
189
+ assert failure.startswith("Judge: evaluation failed")
190
+ assert "no judge mock for criteria 'unmocked'" in failure
191
+
192
+
193
+ # --- scenarios ---------------------------------------------------------------------------------------
194
+
195
+ def test_scenario_runs_each_item_with_shared_and_per_item_assertions():
196
+ async def agent(ctx):
197
+ async def reply():
198
+ if ctx.input == "crash":
199
+ raise RuntimeError("agent crashed")
200
+ return f"echo: {ctx.input}"
201
+ return await ctx.track_step_async("reply", ctx.input, reply)
202
+
203
+ result = asyncio.run(make_test_sdk().run_scenario_async("echo", agent, [TraceAssert.step_exists("reply")], [
204
+ ScenarioItem(name="ok", input="hi", assertions=[TraceAssert.output_contains("echo: hi")]),
205
+ ScenarioItem(name="wrong output", input="bye", assertions=[TraceAssert.output_contains("echo: hi")]),
206
+ ScenarioItem(name="crash", input="crash"),
207
+ ]))
208
+
209
+ assert [r.passed for r in result.results] == [True, False, False]
210
+ assert result.pass_count == 1
211
+ assert any("agent crashed" in f for f in result.results[2].failures)
212
+
213
+
214
+ if __name__ == "__main__":
215
+ tests = [f for name, f in sorted(globals().items()) if name.startswith("test_")]
216
+ for t in tests:
217
+ t()
218
+ print(f"python sdk tests: {len(tests)} passed")