veval-sdk 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
veval/__init__.py ADDED
@@ -0,0 +1,47 @@
1
+ from ._options import VevalOptions
2
+ from ._step import Step, StepHandle
3
+ from ._context import VevalExecutionContext
4
+ from ._tracing import (
5
+ TraceData,
6
+ StepData,
7
+ ReplayOptions,
8
+ ReplayResult,
9
+ )
10
+ from ._snapshots import (
11
+ SnapshotStep,
12
+ SnapshotData,
13
+ SnapshotOptions,
14
+ SnapshotChange,
15
+ SnapshotAlignmentRow,
16
+ SnapshotDiff,
17
+ compare_snapshots,
18
+ )
19
+ from ._assertions import ITraceAssertion, TraceAssert
20
+ from ._scenarios import ScenarioItem, ItemRunResult, ScenarioRunResult
21
+ from ._sdk import VevalSdk
22
+ from ._test_sdk import VevalTestSdk
23
+
24
+ __all__ = [
25
+ "VevalOptions",
26
+ "Step",
27
+ "StepHandle",
28
+ "VevalExecutionContext",
29
+ "TraceData",
30
+ "StepData",
31
+ "SnapshotStep",
32
+ "SnapshotData",
33
+ "SnapshotOptions",
34
+ "SnapshotChange",
35
+ "SnapshotAlignmentRow",
36
+ "SnapshotDiff",
37
+ "compare_snapshots",
38
+ "ReplayOptions",
39
+ "ReplayResult",
40
+ "ITraceAssertion",
41
+ "TraceAssert",
42
+ "ScenarioItem",
43
+ "ItemRunResult",
44
+ "ScenarioRunResult",
45
+ "VevalSdk",
46
+ "VevalTestSdk",
47
+ ]
veval/_assertions.py ADDED
@@ -0,0 +1,141 @@
1
+ from __future__ import annotations
2
+ from typing import Any, Optional, TYPE_CHECKING
3
+ from ._step import Step
4
+ from ._snapshots import SnapshotOptions, compare_snapshots
5
+
6
+ if TYPE_CHECKING:
7
+ from ._context import VevalExecutionContext
8
+
9
+
10
+ class ITraceAssertion:
11
+ async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
12
+ raise NotImplementedError
13
+
14
+
15
+ def _flatten_steps(steps: list[Step]) -> list[Step]:
16
+ result: list[Step] = []
17
+ for s in steps:
18
+ result.append(s)
19
+ result.extend(_flatten_steps(s.children))
20
+ return result
21
+
22
+
23
+ class TraceAssert:
24
+ @staticmethod
25
+ def max_steps(max: int) -> ITraceAssertion:
26
+ class _Assertion(ITraceAssertion):
27
+ async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
28
+ count = len(_flatten_steps(ctx.steps))
29
+ return f"MaxSteps: expected at most {max} steps, got {count}" if count > max else None
30
+ return _Assertion()
31
+
32
+ @staticmethod
33
+ def no_errors() -> ITraceAssertion:
34
+ class _Assertion(ITraceAssertion):
35
+ async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
36
+ errors = [s.name for s in _flatten_steps(ctx.steps) if s.status == "error"]
37
+ return f"NoErrors: found {len(errors)} step(s) with error status: {', '.join(errors)}" if errors else None
38
+ return _Assertion()
39
+
40
+ @staticmethod
41
+ def step_exists(step_name: str) -> ITraceAssertion:
42
+ class _Assertion(ITraceAssertion):
43
+ async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
44
+ found = any(s.name == step_name for s in _flatten_steps(ctx.steps))
45
+ return None if found else f"StepExists: step '{step_name}' not found"
46
+ return _Assertion()
47
+
48
+ @staticmethod
49
+ def max_cost(max_cost: float) -> ITraceAssertion:
50
+ class _Assertion(ITraceAssertion):
51
+ async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
52
+ total = sum(s.cost_usd or 0 for s in _flatten_steps(ctx.steps))
53
+ return f"MaxCost: expected at most {max_cost}, got {total}" if total > max_cost else None
54
+ return _Assertion()
55
+
56
+ @staticmethod
57
+ def max_duration(max_ms: int) -> ITraceAssertion:
58
+ class _Assertion(ITraceAssertion):
59
+ async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
60
+ total = sum(s.duration_ms or 0 for s in _flatten_steps(ctx.steps))
61
+ return f"MaxDuration: expected at most {max_ms}ms, got {total}ms" if total > max_ms else None
62
+ return _Assertion()
63
+
64
+ @staticmethod
65
+ def output_contains(expected: str) -> ITraceAssertion:
66
+ class _Assertion(ITraceAssertion):
67
+ async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
68
+ found = any(
69
+ expected in str(s.output)
70
+ for s in _flatten_steps(ctx.steps)
71
+ if s.output is not None
72
+ )
73
+ return None if found else f"OutputContains: no step output contains '{expected}'"
74
+ return _Assertion()
75
+
76
+ @staticmethod
77
+ def tool_called(tool_name: str) -> ITraceAssertion:
78
+ class _Assertion(ITraceAssertion):
79
+ async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
80
+ found = any(
81
+ s.type == "tool" and s.name == tool_name
82
+ for s in _flatten_steps(ctx.steps)
83
+ )
84
+ return None if found else f"ToolCalled: no tool step named '{tool_name}' was found"
85
+ return _Assertion()
86
+
87
+ @staticmethod
88
+ def judge(
89
+ veval: Any,
90
+ criteria: str,
91
+ model: Optional[str] = None,
92
+ threshold: Optional[float] = None,
93
+ reference_output: Optional[Any] = None,
94
+ samples: Optional[int] = None,
95
+ ) -> ITraceAssertion:
96
+ """Scores the trace's output against a rubric using an LLM judge, evaluated server-side."""
97
+ class _Assertion(ITraceAssertion):
98
+ async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
99
+ try:
100
+ result = await veval.judge_async(
101
+ criteria, ctx, model=model, threshold=threshold,
102
+ reference_output=reference_output, samples=samples,
103
+ )
104
+ except Exception as ex:
105
+ # Never let a network/API failure during judging silently pass a test.
106
+ return f"Judge: evaluation failed — {ex}"
107
+
108
+ ctx.record_judgment(criteria, result["score"], result["passed"], result["reasoning"])
109
+
110
+ if result["passed"]:
111
+ return None
112
+ return f"Judge: {criteria} (score {result['score']:.2f}) — {result['reasoning']}"
113
+ return _Assertion()
114
+
115
+ @staticmethod
116
+ def matches_snapshot(
117
+ snapshot_or_sdk: Any,
118
+ snapshot_name: Optional[str] = None,
119
+ options: Optional[SnapshotOptions] = None,
120
+ ) -> ITraceAssertion:
121
+ """
122
+ Fails when the run's step sequence or step inputs differ from the snapshot — e.g. a changed
123
+ prompt, a dropped or repeated call. Pass a SnapshotData, or (sdk, name) to load a stored
124
+ baseline; a missing baseline fails the assertion.
125
+ """
126
+ by_name = snapshot_name is not None
127
+
128
+ class _Assertion(ITraceAssertion):
129
+ async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
130
+ name = snapshot_name if by_name else snapshot_or_sdk.name
131
+ try:
132
+ snapshot = await snapshot_or_sdk.get_snapshot_async(snapshot_name) if by_name else snapshot_or_sdk
133
+ except Exception as ex:
134
+ # A baseline we couldn't load must fail loudly, never pass vacuously.
135
+ return f"MatchesSnapshot: could not load snapshot '{name}' — {ex}"
136
+ if snapshot is None:
137
+ return f"MatchesSnapshot: no snapshot named '{name}'. Save one with save_snapshot_async first."
138
+
139
+ diff = compare_snapshots(snapshot, ctx, options)
140
+ return "MatchesSnapshot: " + diff.summary(name or snapshot.name) if diff.has_changes else None
141
+ return _Assertion()
veval/_context.py ADDED
@@ -0,0 +1,104 @@
1
+ from __future__ import annotations
2
+ import inspect
3
+ from collections import deque
4
+ from typing import Any, Callable, Awaitable, Optional, TypeVar, TYPE_CHECKING
5
+ from ._step import Step, StepHandle
6
+
7
+ if TYPE_CHECKING:
8
+ from ._tracing import TraceData
9
+
10
+ T = TypeVar("T")
11
+
12
+
13
+ class VevalExecutionContext:
14
+ def __init__(self, trace_id: str, input: Any = None):
15
+ self.trace_id = trace_id
16
+ self.input = input
17
+ self._steps: list[Step] = []
18
+ self._metadata: dict[str, Any] = {}
19
+ self._judgments: list[dict] = []
20
+ self._mock_outputs: Optional[dict[str, deque]] = None
21
+ self._strict_mock_mode = False
22
+
23
+ @property
24
+ def steps(self) -> list[Step]:
25
+ return self._steps
26
+
27
+ @property
28
+ def trace_meta(self) -> dict[str, Any]:
29
+ return self._metadata
30
+
31
+ @property
32
+ def judgments(self) -> list[dict]:
33
+ return self._judgments
34
+
35
+ def record_judgment(self, criteria: str, score: float, passed: bool, reasoning: str) -> None:
36
+ self._judgments.append({
37
+ "criteria": criteria,
38
+ "score": score,
39
+ "passed": passed,
40
+ "reasoning": reasoning,
41
+ })
42
+
43
+ async def track_step_async(
44
+ self,
45
+ name: str,
46
+ input: Any,
47
+ step: Callable[..., Awaitable[T]],
48
+ ) -> T:
49
+ if self._mock_outputs is not None and self._strict_mock_mode:
50
+ q = self._mock_outputs.get(name)
51
+ if not q:
52
+ available = ", ".join(self._mock_outputs.keys())
53
+ raise RuntimeError(
54
+ f"Replay mode: no mock output for step '{name}'. "
55
+ f"Available steps: {available}. "
56
+ "This would have made a real LLM call."
57
+ )
58
+
59
+ if self._mock_outputs is not None:
60
+ q = self._mock_outputs.get(name)
61
+ if q:
62
+ recorded = q.popleft()
63
+ mock = recorded.output
64
+ s = Step(name)
65
+ s.input = input
66
+ # Keep the recorded type so tool_called and snapshot types behave the same as in the original run.
67
+ s.type = recorded.type or "custom"
68
+ s.metadata["_source"] = "replay"
69
+ s.complete(mock)
70
+ self._steps.append(s)
71
+ return mock # type: ignore[return-value]
72
+
73
+ s = Step(name)
74
+ s.input = input
75
+ self._steps.append(s)
76
+ handle = StepHandle(s)
77
+
78
+ try:
79
+ sig = inspect.signature(step)
80
+ if len(sig.parameters) > 0:
81
+ result = await step(handle)
82
+ else:
83
+ result = await step() # type: ignore[call-arg]
84
+ s.complete(result)
85
+ return result
86
+ except Exception as ex:
87
+ s.fail(str(ex))
88
+ raise
89
+
90
+ def set_metadata(self, key: str, value: Any) -> None:
91
+ self._metadata[key] = value
92
+
93
+ def load_mock_outputs(self, trace: TraceData, strict: bool = True) -> None:
94
+ if not trace.steps:
95
+ raise RuntimeError(
96
+ f"Replay trace '{trace.trace_id}' has no steps. Cannot mock LLM calls. "
97
+ "Ensure the trace was recorded with steps before using it for replay."
98
+ )
99
+ self._mock_outputs = {}
100
+ self._strict_mock_mode = strict
101
+ for step in trace.steps:
102
+ if step.name not in self._mock_outputs:
103
+ self._mock_outputs[step.name] = deque()
104
+ self._mock_outputs[step.name].append(step)
veval/_http_client.py ADDED
@@ -0,0 +1,121 @@
1
+ from __future__ import annotations
2
+ import json
3
+ from typing import Any, Optional
4
+ from urllib.parse import quote
5
+ from urllib.request import Request, urlopen
6
+ from urllib.error import HTTPError, URLError
7
+ from ._tracing import TraceData
8
+ from ._snapshots import SnapshotData
9
+
10
+
11
+ class VevalHttpClient:
12
+ def __init__(self, api_key: str, endpoint: str):
13
+ self._api_key = api_key
14
+ self._endpoint = endpoint.rstrip("/")
15
+
16
+ def _headers(self) -> dict[str, str]:
17
+ return {
18
+ "Authorization": f"Bearer {self._api_key}",
19
+ "Content-Type": "application/json",
20
+ }
21
+
22
+ async def send_trace_async(self, payload: Any) -> None:
23
+ try:
24
+ import asyncio
25
+ await asyncio.get_event_loop().run_in_executor(
26
+ None, self._post, f"{self._endpoint}/v1/traces", payload
27
+ )
28
+ except Exception:
29
+ pass
30
+
31
+ async def get_trace_async(self, trace_id: str) -> Optional[TraceData]:
32
+ try:
33
+ import asyncio
34
+ data = await asyncio.get_event_loop().run_in_executor(
35
+ None, self._get, f"{self._endpoint}/v1/traces/{trace_id}"
36
+ )
37
+ return TraceData.from_dict(data) if data else None
38
+ except Exception:
39
+ return None
40
+
41
+ async def get_scenario_items_async(self, scenario_name: str) -> list[dict]:
42
+ try:
43
+ import asyncio
44
+ encoded = quote(scenario_name, safe="")
45
+ data = await asyncio.get_event_loop().run_in_executor(
46
+ None, self._get, f"{self._endpoint}/v1/scenarios/{encoded}/items"
47
+ )
48
+ return data.get("items", []) if data else []
49
+ except Exception:
50
+ return []
51
+
52
+ async def post_scenario_run_async(self, scenario_name: str, payload: Any) -> None:
53
+ try:
54
+ import asyncio
55
+ encoded = quote(scenario_name, safe="")
56
+ await asyncio.get_event_loop().run_in_executor(
57
+ None, self._post, f"{self._endpoint}/v1/scenarios/{encoded}/runs", payload
58
+ )
59
+ except Exception:
60
+ pass
61
+
62
+ async def judge_async(self, payload: Any) -> dict:
63
+ # Unlike the other calls on this client, judge_async deliberately does not swallow
64
+ # failures — a network/API error here must surface as an assertion failure, not a
65
+ # silent pass. Callers are expected to catch.
66
+ import asyncio
67
+ return await asyncio.get_event_loop().run_in_executor(
68
+ None, self._post_json, f"{self._endpoint}/v1/judge", payload
69
+ )
70
+
71
+ async def create_snapshot_async(self, payload: Any) -> SnapshotData:
72
+ # Raises on failure — a baseline that silently didn't save would make every later
73
+ # comparison fail, or compare against a stale one.
74
+ import asyncio
75
+ data = await asyncio.get_event_loop().run_in_executor(
76
+ None, self._post_json, f"{self._endpoint}/v1/snapshots", payload
77
+ )
78
+ return SnapshotData.from_dict(data)
79
+
80
+ async def get_snapshot_async(self, name: str) -> Optional[SnapshotData]:
81
+ # Returns None only when no baseline exists (404); any other failure raises, so a network
82
+ # error can never look like "nothing to compare against".
83
+ import asyncio
84
+ encoded = quote(name, safe="")
85
+ data = await asyncio.get_event_loop().run_in_executor(
86
+ None, self._get_or_none_on_404, f"{self._endpoint}/v1/snapshots/{encoded}"
87
+ )
88
+ return SnapshotData.from_dict(data) if data is not None else None
89
+
90
+ def _get_or_none_on_404(self, url: str) -> Optional[dict]:
91
+ req = Request(url, headers=self._headers(), method="GET")
92
+ try:
93
+ with urlopen(req, timeout=10) as resp:
94
+ return json.loads(resp.read().decode("utf-8"))
95
+ except HTTPError as ex:
96
+ if ex.code == 404:
97
+ return None
98
+ raise
99
+
100
+ def _post_json(self, url: str, payload: Any) -> dict:
101
+ body = json.dumps(payload, default=str).encode("utf-8")
102
+ req = Request(url, data=body, headers=self._headers(), method="POST")
103
+ with urlopen(req, timeout=30) as resp:
104
+ return json.loads(resp.read().decode("utf-8"))
105
+
106
+ def _post(self, url: str, payload: Any) -> None:
107
+ body = json.dumps(payload, default=str).encode("utf-8")
108
+ req = Request(url, data=body, headers=self._headers(), method="POST")
109
+ try:
110
+ with urlopen(req, timeout=10):
111
+ pass
112
+ except URLError:
113
+ pass
114
+
115
+ def _get(self, url: str) -> Optional[dict]:
116
+ req = Request(url, headers=self._headers(), method="GET")
117
+ try:
118
+ with urlopen(req, timeout=10) as resp:
119
+ return json.loads(resp.read().decode("utf-8"))
120
+ except URLError:
121
+ return None
veval/_options.py ADDED
@@ -0,0 +1,29 @@
1
+ import os
2
+ import warnings
3
+
4
+
5
+ class VevalOptions:
6
+ def __init__(
7
+ self,
8
+ api_key: str = "",
9
+ project_id: str = "",
10
+ flush_interval_ms: int = 5000,
11
+ flush_batch_size: int = 50,
12
+ ):
13
+ # The API key also determines the workspace.
14
+ self.api_key = api_key
15
+ if project_id:
16
+ warnings.warn(
17
+ "VevalOptions.project_id is ignored: the API key determines the workspace. "
18
+ "It will be removed in a future version.",
19
+ DeprecationWarning,
20
+ stacklevel=2,
21
+ )
22
+ self.project_id = project_id
23
+ # Veval's own hosted API, which enforces billing/quota (test-run limits, judge credits, etc.) —
24
+ # not a self-hosted override for SDK consumers, so it isn't a public option.
25
+ # VEVAL_INTERNAL_ENDPOINT is an undocumented escape hatch for Veval's own local development
26
+ # against a non-production API instance.
27
+ self._endpoint = os.environ.get("VEVAL_INTERNAL_ENDPOINT", "https://api.veval.dev")
28
+ self.flush_interval_ms = flush_interval_ms
29
+ self.flush_batch_size = flush_batch_size
veval/_scenarios.py ADDED
@@ -0,0 +1,43 @@
1
+ from __future__ import annotations
2
+ from dataclasses import dataclass, field
3
+ from typing import Any, Optional, TYPE_CHECKING
4
+ from ._assertions import ITraceAssertion
5
+
6
+ if TYPE_CHECKING:
7
+ from ._context import VevalExecutionContext
8
+
9
+
10
+ @dataclass
11
+ class ScenarioItem:
12
+ name: Optional[str] = None
13
+ trace_id: Optional[str] = None
14
+ input: Any = None
15
+ assertions: list[ITraceAssertion] = field(default_factory=list)
16
+
17
+
18
+ @dataclass
19
+ class ItemRunResult:
20
+ item: ScenarioItem = field(default_factory=ScenarioItem)
21
+ failures: list[str] = field(default_factory=list)
22
+ context: Optional[VevalExecutionContext] = None
23
+
24
+ @property
25
+ def passed(self) -> bool:
26
+ return len(self.failures) == 0
27
+
28
+
29
+ @dataclass
30
+ class ScenarioRunResult:
31
+ results: list[ItemRunResult] = field(default_factory=list)
32
+
33
+ @property
34
+ def passed(self) -> bool:
35
+ return all(r.passed for r in self.results)
36
+
37
+ @property
38
+ def pass_count(self) -> int:
39
+ return sum(1 for r in self.results if r.passed)
40
+
41
+ @property
42
+ def fail_count(self) -> int:
43
+ return sum(1 for r in self.results if not r.passed)