veval-sdk 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- veval/__init__.py +47 -0
- veval/_assertions.py +141 -0
- veval/_context.py +104 -0
- veval/_http_client.py +121 -0
- veval/_options.py +29 -0
- veval/_scenarios.py +43 -0
- veval/_sdk.py +332 -0
- veval/_snapshots.py +431 -0
- veval/_step.py +68 -0
- veval/_test_sdk.py +216 -0
- veval/_tracing.py +121 -0
- veval_sdk-1.1.0.dist-info/METADATA +208 -0
- veval_sdk-1.1.0.dist-info/RECORD +15 -0
- veval_sdk-1.1.0.dist-info/WHEEL +4 -0
- veval_sdk-1.1.0.dist-info/licenses/LICENSE +21 -0
veval/__init__.py
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
from ._options import VevalOptions
|
|
2
|
+
from ._step import Step, StepHandle
|
|
3
|
+
from ._context import VevalExecutionContext
|
|
4
|
+
from ._tracing import (
|
|
5
|
+
TraceData,
|
|
6
|
+
StepData,
|
|
7
|
+
ReplayOptions,
|
|
8
|
+
ReplayResult,
|
|
9
|
+
)
|
|
10
|
+
from ._snapshots import (
|
|
11
|
+
SnapshotStep,
|
|
12
|
+
SnapshotData,
|
|
13
|
+
SnapshotOptions,
|
|
14
|
+
SnapshotChange,
|
|
15
|
+
SnapshotAlignmentRow,
|
|
16
|
+
SnapshotDiff,
|
|
17
|
+
compare_snapshots,
|
|
18
|
+
)
|
|
19
|
+
from ._assertions import ITraceAssertion, TraceAssert
|
|
20
|
+
from ._scenarios import ScenarioItem, ItemRunResult, ScenarioRunResult
|
|
21
|
+
from ._sdk import VevalSdk
|
|
22
|
+
from ._test_sdk import VevalTestSdk
|
|
23
|
+
|
|
24
|
+
__all__ = [
|
|
25
|
+
"VevalOptions",
|
|
26
|
+
"Step",
|
|
27
|
+
"StepHandle",
|
|
28
|
+
"VevalExecutionContext",
|
|
29
|
+
"TraceData",
|
|
30
|
+
"StepData",
|
|
31
|
+
"SnapshotStep",
|
|
32
|
+
"SnapshotData",
|
|
33
|
+
"SnapshotOptions",
|
|
34
|
+
"SnapshotChange",
|
|
35
|
+
"SnapshotAlignmentRow",
|
|
36
|
+
"SnapshotDiff",
|
|
37
|
+
"compare_snapshots",
|
|
38
|
+
"ReplayOptions",
|
|
39
|
+
"ReplayResult",
|
|
40
|
+
"ITraceAssertion",
|
|
41
|
+
"TraceAssert",
|
|
42
|
+
"ScenarioItem",
|
|
43
|
+
"ItemRunResult",
|
|
44
|
+
"ScenarioRunResult",
|
|
45
|
+
"VevalSdk",
|
|
46
|
+
"VevalTestSdk",
|
|
47
|
+
]
|
veval/_assertions.py
ADDED
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
from typing import Any, Optional, TYPE_CHECKING
|
|
3
|
+
from ._step import Step
|
|
4
|
+
from ._snapshots import SnapshotOptions, compare_snapshots
|
|
5
|
+
|
|
6
|
+
if TYPE_CHECKING:
|
|
7
|
+
from ._context import VevalExecutionContext
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ITraceAssertion:
|
|
11
|
+
async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
|
|
12
|
+
raise NotImplementedError
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _flatten_steps(steps: list[Step]) -> list[Step]:
|
|
16
|
+
result: list[Step] = []
|
|
17
|
+
for s in steps:
|
|
18
|
+
result.append(s)
|
|
19
|
+
result.extend(_flatten_steps(s.children))
|
|
20
|
+
return result
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class TraceAssert:
|
|
24
|
+
@staticmethod
|
|
25
|
+
def max_steps(max: int) -> ITraceAssertion:
|
|
26
|
+
class _Assertion(ITraceAssertion):
|
|
27
|
+
async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
|
|
28
|
+
count = len(_flatten_steps(ctx.steps))
|
|
29
|
+
return f"MaxSteps: expected at most {max} steps, got {count}" if count > max else None
|
|
30
|
+
return _Assertion()
|
|
31
|
+
|
|
32
|
+
@staticmethod
|
|
33
|
+
def no_errors() -> ITraceAssertion:
|
|
34
|
+
class _Assertion(ITraceAssertion):
|
|
35
|
+
async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
|
|
36
|
+
errors = [s.name for s in _flatten_steps(ctx.steps) if s.status == "error"]
|
|
37
|
+
return f"NoErrors: found {len(errors)} step(s) with error status: {', '.join(errors)}" if errors else None
|
|
38
|
+
return _Assertion()
|
|
39
|
+
|
|
40
|
+
@staticmethod
|
|
41
|
+
def step_exists(step_name: str) -> ITraceAssertion:
|
|
42
|
+
class _Assertion(ITraceAssertion):
|
|
43
|
+
async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
|
|
44
|
+
found = any(s.name == step_name for s in _flatten_steps(ctx.steps))
|
|
45
|
+
return None if found else f"StepExists: step '{step_name}' not found"
|
|
46
|
+
return _Assertion()
|
|
47
|
+
|
|
48
|
+
@staticmethod
|
|
49
|
+
def max_cost(max_cost: float) -> ITraceAssertion:
|
|
50
|
+
class _Assertion(ITraceAssertion):
|
|
51
|
+
async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
|
|
52
|
+
total = sum(s.cost_usd or 0 for s in _flatten_steps(ctx.steps))
|
|
53
|
+
return f"MaxCost: expected at most {max_cost}, got {total}" if total > max_cost else None
|
|
54
|
+
return _Assertion()
|
|
55
|
+
|
|
56
|
+
@staticmethod
|
|
57
|
+
def max_duration(max_ms: int) -> ITraceAssertion:
|
|
58
|
+
class _Assertion(ITraceAssertion):
|
|
59
|
+
async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
|
|
60
|
+
total = sum(s.duration_ms or 0 for s in _flatten_steps(ctx.steps))
|
|
61
|
+
return f"MaxDuration: expected at most {max_ms}ms, got {total}ms" if total > max_ms else None
|
|
62
|
+
return _Assertion()
|
|
63
|
+
|
|
64
|
+
@staticmethod
|
|
65
|
+
def output_contains(expected: str) -> ITraceAssertion:
|
|
66
|
+
class _Assertion(ITraceAssertion):
|
|
67
|
+
async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
|
|
68
|
+
found = any(
|
|
69
|
+
expected in str(s.output)
|
|
70
|
+
for s in _flatten_steps(ctx.steps)
|
|
71
|
+
if s.output is not None
|
|
72
|
+
)
|
|
73
|
+
return None if found else f"OutputContains: no step output contains '{expected}'"
|
|
74
|
+
return _Assertion()
|
|
75
|
+
|
|
76
|
+
@staticmethod
|
|
77
|
+
def tool_called(tool_name: str) -> ITraceAssertion:
|
|
78
|
+
class _Assertion(ITraceAssertion):
|
|
79
|
+
async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
|
|
80
|
+
found = any(
|
|
81
|
+
s.type == "tool" and s.name == tool_name
|
|
82
|
+
for s in _flatten_steps(ctx.steps)
|
|
83
|
+
)
|
|
84
|
+
return None if found else f"ToolCalled: no tool step named '{tool_name}' was found"
|
|
85
|
+
return _Assertion()
|
|
86
|
+
|
|
87
|
+
@staticmethod
|
|
88
|
+
def judge(
|
|
89
|
+
veval: Any,
|
|
90
|
+
criteria: str,
|
|
91
|
+
model: Optional[str] = None,
|
|
92
|
+
threshold: Optional[float] = None,
|
|
93
|
+
reference_output: Optional[Any] = None,
|
|
94
|
+
samples: Optional[int] = None,
|
|
95
|
+
) -> ITraceAssertion:
|
|
96
|
+
"""Scores the trace's output against a rubric using an LLM judge, evaluated server-side."""
|
|
97
|
+
class _Assertion(ITraceAssertion):
|
|
98
|
+
async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
|
|
99
|
+
try:
|
|
100
|
+
result = await veval.judge_async(
|
|
101
|
+
criteria, ctx, model=model, threshold=threshold,
|
|
102
|
+
reference_output=reference_output, samples=samples,
|
|
103
|
+
)
|
|
104
|
+
except Exception as ex:
|
|
105
|
+
# Never let a network/API failure during judging silently pass a test.
|
|
106
|
+
return f"Judge: evaluation failed — {ex}"
|
|
107
|
+
|
|
108
|
+
ctx.record_judgment(criteria, result["score"], result["passed"], result["reasoning"])
|
|
109
|
+
|
|
110
|
+
if result["passed"]:
|
|
111
|
+
return None
|
|
112
|
+
return f"Judge: {criteria} (score {result['score']:.2f}) — {result['reasoning']}"
|
|
113
|
+
return _Assertion()
|
|
114
|
+
|
|
115
|
+
@staticmethod
|
|
116
|
+
def matches_snapshot(
|
|
117
|
+
snapshot_or_sdk: Any,
|
|
118
|
+
snapshot_name: Optional[str] = None,
|
|
119
|
+
options: Optional[SnapshotOptions] = None,
|
|
120
|
+
) -> ITraceAssertion:
|
|
121
|
+
"""
|
|
122
|
+
Fails when the run's step sequence or step inputs differ from the snapshot — e.g. a changed
|
|
123
|
+
prompt, a dropped or repeated call. Pass a SnapshotData, or (sdk, name) to load a stored
|
|
124
|
+
baseline; a missing baseline fails the assertion.
|
|
125
|
+
"""
|
|
126
|
+
by_name = snapshot_name is not None
|
|
127
|
+
|
|
128
|
+
class _Assertion(ITraceAssertion):
|
|
129
|
+
async def evaluate_async(self, ctx: VevalExecutionContext) -> Optional[str]:
|
|
130
|
+
name = snapshot_name if by_name else snapshot_or_sdk.name
|
|
131
|
+
try:
|
|
132
|
+
snapshot = await snapshot_or_sdk.get_snapshot_async(snapshot_name) if by_name else snapshot_or_sdk
|
|
133
|
+
except Exception as ex:
|
|
134
|
+
# A baseline we couldn't load must fail loudly, never pass vacuously.
|
|
135
|
+
return f"MatchesSnapshot: could not load snapshot '{name}' — {ex}"
|
|
136
|
+
if snapshot is None:
|
|
137
|
+
return f"MatchesSnapshot: no snapshot named '{name}'. Save one with save_snapshot_async first."
|
|
138
|
+
|
|
139
|
+
diff = compare_snapshots(snapshot, ctx, options)
|
|
140
|
+
return "MatchesSnapshot: " + diff.summary(name or snapshot.name) if diff.has_changes else None
|
|
141
|
+
return _Assertion()
|
veval/_context.py
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
import inspect
|
|
3
|
+
from collections import deque
|
|
4
|
+
from typing import Any, Callable, Awaitable, Optional, TypeVar, TYPE_CHECKING
|
|
5
|
+
from ._step import Step, StepHandle
|
|
6
|
+
|
|
7
|
+
if TYPE_CHECKING:
|
|
8
|
+
from ._tracing import TraceData
|
|
9
|
+
|
|
10
|
+
T = TypeVar("T")
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class VevalExecutionContext:
|
|
14
|
+
def __init__(self, trace_id: str, input: Any = None):
|
|
15
|
+
self.trace_id = trace_id
|
|
16
|
+
self.input = input
|
|
17
|
+
self._steps: list[Step] = []
|
|
18
|
+
self._metadata: dict[str, Any] = {}
|
|
19
|
+
self._judgments: list[dict] = []
|
|
20
|
+
self._mock_outputs: Optional[dict[str, deque]] = None
|
|
21
|
+
self._strict_mock_mode = False
|
|
22
|
+
|
|
23
|
+
@property
|
|
24
|
+
def steps(self) -> list[Step]:
|
|
25
|
+
return self._steps
|
|
26
|
+
|
|
27
|
+
@property
|
|
28
|
+
def trace_meta(self) -> dict[str, Any]:
|
|
29
|
+
return self._metadata
|
|
30
|
+
|
|
31
|
+
@property
|
|
32
|
+
def judgments(self) -> list[dict]:
|
|
33
|
+
return self._judgments
|
|
34
|
+
|
|
35
|
+
def record_judgment(self, criteria: str, score: float, passed: bool, reasoning: str) -> None:
|
|
36
|
+
self._judgments.append({
|
|
37
|
+
"criteria": criteria,
|
|
38
|
+
"score": score,
|
|
39
|
+
"passed": passed,
|
|
40
|
+
"reasoning": reasoning,
|
|
41
|
+
})
|
|
42
|
+
|
|
43
|
+
async def track_step_async(
|
|
44
|
+
self,
|
|
45
|
+
name: str,
|
|
46
|
+
input: Any,
|
|
47
|
+
step: Callable[..., Awaitable[T]],
|
|
48
|
+
) -> T:
|
|
49
|
+
if self._mock_outputs is not None and self._strict_mock_mode:
|
|
50
|
+
q = self._mock_outputs.get(name)
|
|
51
|
+
if not q:
|
|
52
|
+
available = ", ".join(self._mock_outputs.keys())
|
|
53
|
+
raise RuntimeError(
|
|
54
|
+
f"Replay mode: no mock output for step '{name}'. "
|
|
55
|
+
f"Available steps: {available}. "
|
|
56
|
+
"This would have made a real LLM call."
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
if self._mock_outputs is not None:
|
|
60
|
+
q = self._mock_outputs.get(name)
|
|
61
|
+
if q:
|
|
62
|
+
recorded = q.popleft()
|
|
63
|
+
mock = recorded.output
|
|
64
|
+
s = Step(name)
|
|
65
|
+
s.input = input
|
|
66
|
+
# Keep the recorded type so tool_called and snapshot types behave the same as in the original run.
|
|
67
|
+
s.type = recorded.type or "custom"
|
|
68
|
+
s.metadata["_source"] = "replay"
|
|
69
|
+
s.complete(mock)
|
|
70
|
+
self._steps.append(s)
|
|
71
|
+
return mock # type: ignore[return-value]
|
|
72
|
+
|
|
73
|
+
s = Step(name)
|
|
74
|
+
s.input = input
|
|
75
|
+
self._steps.append(s)
|
|
76
|
+
handle = StepHandle(s)
|
|
77
|
+
|
|
78
|
+
try:
|
|
79
|
+
sig = inspect.signature(step)
|
|
80
|
+
if len(sig.parameters) > 0:
|
|
81
|
+
result = await step(handle)
|
|
82
|
+
else:
|
|
83
|
+
result = await step() # type: ignore[call-arg]
|
|
84
|
+
s.complete(result)
|
|
85
|
+
return result
|
|
86
|
+
except Exception as ex:
|
|
87
|
+
s.fail(str(ex))
|
|
88
|
+
raise
|
|
89
|
+
|
|
90
|
+
def set_metadata(self, key: str, value: Any) -> None:
|
|
91
|
+
self._metadata[key] = value
|
|
92
|
+
|
|
93
|
+
def load_mock_outputs(self, trace: TraceData, strict: bool = True) -> None:
|
|
94
|
+
if not trace.steps:
|
|
95
|
+
raise RuntimeError(
|
|
96
|
+
f"Replay trace '{trace.trace_id}' has no steps. Cannot mock LLM calls. "
|
|
97
|
+
"Ensure the trace was recorded with steps before using it for replay."
|
|
98
|
+
)
|
|
99
|
+
self._mock_outputs = {}
|
|
100
|
+
self._strict_mock_mode = strict
|
|
101
|
+
for step in trace.steps:
|
|
102
|
+
if step.name not in self._mock_outputs:
|
|
103
|
+
self._mock_outputs[step.name] = deque()
|
|
104
|
+
self._mock_outputs[step.name].append(step)
|
veval/_http_client.py
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
import json
|
|
3
|
+
from typing import Any, Optional
|
|
4
|
+
from urllib.parse import quote
|
|
5
|
+
from urllib.request import Request, urlopen
|
|
6
|
+
from urllib.error import HTTPError, URLError
|
|
7
|
+
from ._tracing import TraceData
|
|
8
|
+
from ._snapshots import SnapshotData
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class VevalHttpClient:
|
|
12
|
+
def __init__(self, api_key: str, endpoint: str):
|
|
13
|
+
self._api_key = api_key
|
|
14
|
+
self._endpoint = endpoint.rstrip("/")
|
|
15
|
+
|
|
16
|
+
def _headers(self) -> dict[str, str]:
|
|
17
|
+
return {
|
|
18
|
+
"Authorization": f"Bearer {self._api_key}",
|
|
19
|
+
"Content-Type": "application/json",
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
async def send_trace_async(self, payload: Any) -> None:
|
|
23
|
+
try:
|
|
24
|
+
import asyncio
|
|
25
|
+
await asyncio.get_event_loop().run_in_executor(
|
|
26
|
+
None, self._post, f"{self._endpoint}/v1/traces", payload
|
|
27
|
+
)
|
|
28
|
+
except Exception:
|
|
29
|
+
pass
|
|
30
|
+
|
|
31
|
+
async def get_trace_async(self, trace_id: str) -> Optional[TraceData]:
|
|
32
|
+
try:
|
|
33
|
+
import asyncio
|
|
34
|
+
data = await asyncio.get_event_loop().run_in_executor(
|
|
35
|
+
None, self._get, f"{self._endpoint}/v1/traces/{trace_id}"
|
|
36
|
+
)
|
|
37
|
+
return TraceData.from_dict(data) if data else None
|
|
38
|
+
except Exception:
|
|
39
|
+
return None
|
|
40
|
+
|
|
41
|
+
async def get_scenario_items_async(self, scenario_name: str) -> list[dict]:
|
|
42
|
+
try:
|
|
43
|
+
import asyncio
|
|
44
|
+
encoded = quote(scenario_name, safe="")
|
|
45
|
+
data = await asyncio.get_event_loop().run_in_executor(
|
|
46
|
+
None, self._get, f"{self._endpoint}/v1/scenarios/{encoded}/items"
|
|
47
|
+
)
|
|
48
|
+
return data.get("items", []) if data else []
|
|
49
|
+
except Exception:
|
|
50
|
+
return []
|
|
51
|
+
|
|
52
|
+
async def post_scenario_run_async(self, scenario_name: str, payload: Any) -> None:
|
|
53
|
+
try:
|
|
54
|
+
import asyncio
|
|
55
|
+
encoded = quote(scenario_name, safe="")
|
|
56
|
+
await asyncio.get_event_loop().run_in_executor(
|
|
57
|
+
None, self._post, f"{self._endpoint}/v1/scenarios/{encoded}/runs", payload
|
|
58
|
+
)
|
|
59
|
+
except Exception:
|
|
60
|
+
pass
|
|
61
|
+
|
|
62
|
+
async def judge_async(self, payload: Any) -> dict:
|
|
63
|
+
# Unlike the other calls on this client, judge_async deliberately does not swallow
|
|
64
|
+
# failures — a network/API error here must surface as an assertion failure, not a
|
|
65
|
+
# silent pass. Callers are expected to catch.
|
|
66
|
+
import asyncio
|
|
67
|
+
return await asyncio.get_event_loop().run_in_executor(
|
|
68
|
+
None, self._post_json, f"{self._endpoint}/v1/judge", payload
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
async def create_snapshot_async(self, payload: Any) -> SnapshotData:
|
|
72
|
+
# Raises on failure — a baseline that silently didn't save would make every later
|
|
73
|
+
# comparison fail, or compare against a stale one.
|
|
74
|
+
import asyncio
|
|
75
|
+
data = await asyncio.get_event_loop().run_in_executor(
|
|
76
|
+
None, self._post_json, f"{self._endpoint}/v1/snapshots", payload
|
|
77
|
+
)
|
|
78
|
+
return SnapshotData.from_dict(data)
|
|
79
|
+
|
|
80
|
+
async def get_snapshot_async(self, name: str) -> Optional[SnapshotData]:
|
|
81
|
+
# Returns None only when no baseline exists (404); any other failure raises, so a network
|
|
82
|
+
# error can never look like "nothing to compare against".
|
|
83
|
+
import asyncio
|
|
84
|
+
encoded = quote(name, safe="")
|
|
85
|
+
data = await asyncio.get_event_loop().run_in_executor(
|
|
86
|
+
None, self._get_or_none_on_404, f"{self._endpoint}/v1/snapshots/{encoded}"
|
|
87
|
+
)
|
|
88
|
+
return SnapshotData.from_dict(data) if data is not None else None
|
|
89
|
+
|
|
90
|
+
def _get_or_none_on_404(self, url: str) -> Optional[dict]:
|
|
91
|
+
req = Request(url, headers=self._headers(), method="GET")
|
|
92
|
+
try:
|
|
93
|
+
with urlopen(req, timeout=10) as resp:
|
|
94
|
+
return json.loads(resp.read().decode("utf-8"))
|
|
95
|
+
except HTTPError as ex:
|
|
96
|
+
if ex.code == 404:
|
|
97
|
+
return None
|
|
98
|
+
raise
|
|
99
|
+
|
|
100
|
+
def _post_json(self, url: str, payload: Any) -> dict:
|
|
101
|
+
body = json.dumps(payload, default=str).encode("utf-8")
|
|
102
|
+
req = Request(url, data=body, headers=self._headers(), method="POST")
|
|
103
|
+
with urlopen(req, timeout=30) as resp:
|
|
104
|
+
return json.loads(resp.read().decode("utf-8"))
|
|
105
|
+
|
|
106
|
+
def _post(self, url: str, payload: Any) -> None:
|
|
107
|
+
body = json.dumps(payload, default=str).encode("utf-8")
|
|
108
|
+
req = Request(url, data=body, headers=self._headers(), method="POST")
|
|
109
|
+
try:
|
|
110
|
+
with urlopen(req, timeout=10):
|
|
111
|
+
pass
|
|
112
|
+
except URLError:
|
|
113
|
+
pass
|
|
114
|
+
|
|
115
|
+
def _get(self, url: str) -> Optional[dict]:
|
|
116
|
+
req = Request(url, headers=self._headers(), method="GET")
|
|
117
|
+
try:
|
|
118
|
+
with urlopen(req, timeout=10) as resp:
|
|
119
|
+
return json.loads(resp.read().decode("utf-8"))
|
|
120
|
+
except URLError:
|
|
121
|
+
return None
|
veval/_options.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import warnings
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class VevalOptions:
|
|
6
|
+
def __init__(
|
|
7
|
+
self,
|
|
8
|
+
api_key: str = "",
|
|
9
|
+
project_id: str = "",
|
|
10
|
+
flush_interval_ms: int = 5000,
|
|
11
|
+
flush_batch_size: int = 50,
|
|
12
|
+
):
|
|
13
|
+
# The API key also determines the workspace.
|
|
14
|
+
self.api_key = api_key
|
|
15
|
+
if project_id:
|
|
16
|
+
warnings.warn(
|
|
17
|
+
"VevalOptions.project_id is ignored: the API key determines the workspace. "
|
|
18
|
+
"It will be removed in a future version.",
|
|
19
|
+
DeprecationWarning,
|
|
20
|
+
stacklevel=2,
|
|
21
|
+
)
|
|
22
|
+
self.project_id = project_id
|
|
23
|
+
# Veval's own hosted API, which enforces billing/quota (test-run limits, judge credits, etc.) —
|
|
24
|
+
# not a self-hosted override for SDK consumers, so it isn't a public option.
|
|
25
|
+
# VEVAL_INTERNAL_ENDPOINT is an undocumented escape hatch for Veval's own local development
|
|
26
|
+
# against a non-production API instance.
|
|
27
|
+
self._endpoint = os.environ.get("VEVAL_INTERNAL_ENDPOINT", "https://api.veval.dev")
|
|
28
|
+
self.flush_interval_ms = flush_interval_ms
|
|
29
|
+
self.flush_batch_size = flush_batch_size
|
veval/_scenarios.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
from dataclasses import dataclass, field
|
|
3
|
+
from typing import Any, Optional, TYPE_CHECKING
|
|
4
|
+
from ._assertions import ITraceAssertion
|
|
5
|
+
|
|
6
|
+
if TYPE_CHECKING:
|
|
7
|
+
from ._context import VevalExecutionContext
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass
|
|
11
|
+
class ScenarioItem:
|
|
12
|
+
name: Optional[str] = None
|
|
13
|
+
trace_id: Optional[str] = None
|
|
14
|
+
input: Any = None
|
|
15
|
+
assertions: list[ITraceAssertion] = field(default_factory=list)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass
|
|
19
|
+
class ItemRunResult:
|
|
20
|
+
item: ScenarioItem = field(default_factory=ScenarioItem)
|
|
21
|
+
failures: list[str] = field(default_factory=list)
|
|
22
|
+
context: Optional[VevalExecutionContext] = None
|
|
23
|
+
|
|
24
|
+
@property
|
|
25
|
+
def passed(self) -> bool:
|
|
26
|
+
return len(self.failures) == 0
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class ScenarioRunResult:
|
|
31
|
+
results: list[ItemRunResult] = field(default_factory=list)
|
|
32
|
+
|
|
33
|
+
@property
|
|
34
|
+
def passed(self) -> bool:
|
|
35
|
+
return all(r.passed for r in self.results)
|
|
36
|
+
|
|
37
|
+
@property
|
|
38
|
+
def pass_count(self) -> int:
|
|
39
|
+
return sum(1 for r in self.results if r.passed)
|
|
40
|
+
|
|
41
|
+
@property
|
|
42
|
+
def fail_count(self) -> int:
|
|
43
|
+
return sum(1 for r in self.results if not r.passed)
|