outturn 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
outturn/__init__.py ADDED
@@ -0,0 +1,5 @@
1
+ """outturn: did the agent produce the right outcome? As a number."""
2
+ from .models import Discrepancy, Entry, Kind, Outcome, Scenario, Turn
3
+
4
+ __version__ = "0.1.0"
5
+ __all__ = ["Outcome", "Entry", "Scenario", "Turn", "Discrepancy", "Kind", "__version__"]
outturn/__main__.py ADDED
@@ -0,0 +1,5 @@
1
+ import sys
2
+
3
+ from .cli import main
4
+
5
+ sys.exit(main())
@@ -0,0 +1,68 @@
1
+ """Adapter lookup.
2
+
3
+ Two ways to name an adapter:
4
+
5
+ --adapter demo-booking
6
+ one of the demo agents that ship with outturn
7
+
8
+ --adapter myproject.agents:MyAgent
9
+ anything importable. Your agent lives in your repo, not in this one.
10
+
11
+ The second form is the point. You should never have to edit a file inside
12
+ this package to test your own agent.
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import importlib
17
+
18
+ from .base import Agent
19
+ from .demo_booking import DemoBookingAgent
20
+ from .demo_order import DemoOrderAgent
21
+
22
+ BUILTIN: dict[str, type] = {
23
+ DemoOrderAgent.name: DemoOrderAgent,
24
+ DemoBookingAgent.name: DemoBookingAgent,
25
+ }
26
+
27
+
28
+ class AdapterError(ValueError):
29
+ pass
30
+
31
+
32
+ def _load_path(spec: str) -> type:
33
+ """Import 'package.module:ClassName' or 'package.module.ClassName'."""
34
+ if ":" in spec:
35
+ module_name, _, attr = spec.partition(":")
36
+ else:
37
+ module_name, _, attr = spec.rpartition(".")
38
+ if not module_name or not attr:
39
+ raise AdapterError(
40
+ f"{spec!r} is not a known adapter and does not look like an import path. "
41
+ f"Use 'module:ClassName', or one of: {', '.join(sorted(BUILTIN))}"
42
+ )
43
+ try:
44
+ module = importlib.import_module(module_name)
45
+ except ImportError as e:
46
+ raise AdapterError(f"could not import {module_name!r}: {e}") from e
47
+ try:
48
+ return getattr(module, attr)
49
+ except AttributeError as e:
50
+ raise AdapterError(f"{module_name!r} has no attribute {attr!r}") from e
51
+
52
+
53
+ def get(spec: str) -> Agent:
54
+ """Resolve an adapter name or import path into an instance."""
55
+ cls = BUILTIN.get(spec) or _load_path(spec)
56
+
57
+ agent = cls() if isinstance(cls, type) else cls # a factory function is fine too
58
+
59
+ for method in ("reset", "send", "outcome"):
60
+ if not callable(getattr(agent, method, None)):
61
+ raise AdapterError(
62
+ f"{spec!r} is missing {method}(). An adapter needs reset(), send(text) "
63
+ f"and outcome()."
64
+ )
65
+ return agent
66
+
67
+
68
+ __all__ = ["Agent", "AdapterError", "BUILTIN", "get", "DemoOrderAgent", "DemoBookingAgent"]
@@ -0,0 +1,25 @@
1
+ """The adapter contract. Three methods.
2
+
3
+ reset() start a fresh conversation
4
+ send(text) one caller turn in, the agent's reply out
5
+ outcome() the structured result, right now
6
+
7
+ If your agent cannot hand back a structured outcome, outturn cannot help you,
8
+ and that is itself the finding. State that lives only inside the model's
9
+ context is not inspectable, and what is not inspectable is not testable.
10
+ Move the outcome into code and the tool works.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ from typing import Protocol, runtime_checkable
15
+
16
+ from ..models import Outcome
17
+
18
+
19
+ @runtime_checkable
20
+ class Agent(Protocol):
21
+ name: str
22
+
23
+ def reset(self) -> None: ...
24
+ def send(self, text: str) -> str: ...
25
+ def outcome(self) -> Outcome: ...
@@ -0,0 +1,75 @@
1
+ """A deliberately imperfect appointment booking agent.
2
+
3
+ This one exists to prove the point of the whole rewrite: a booking has no
4
+ line items at all. It is a handful of scalar fields. The same engine handles
5
+ it without a single special case.
6
+
7
+ Its planted flaw is the one real booking agents actually have: it hears the
8
+ service and the day, and forgets to carry the duration across when the
9
+ customer changes their mind mid call.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import random
14
+ import re
15
+
16
+ from ..models import Outcome
17
+
18
+ SERVICES = {
19
+ "deep tissue massage": 60,
20
+ "swedish massage": 60,
21
+ "sports massage": 90,
22
+ "facial": 45,
23
+ }
24
+ _DAYS = ("monday", "tuesday", "wednesday", "thursday", "friday", "saturday", "sunday")
25
+
26
+
27
+ class DemoBookingAgent:
28
+ name = "demo-booking"
29
+
30
+ def __init__(self, forget_duration_rate: float = 0.4) -> None:
31
+ self.forget_duration_rate = forget_duration_rate
32
+ self._state: dict = {}
33
+
34
+ def reset(self) -> None:
35
+ self._state = {"confirmed": False}
36
+
37
+ def send(self, text: str) -> str:
38
+ t = text.lower().strip()
39
+ changed = False
40
+
41
+ for service, minutes in SERVICES.items():
42
+ if service in t or service.split()[0] in t:
43
+ switching = "service" in self._state and self._state["service"] != service
44
+ self._state["service"] = service
45
+ # the planted flaw: on a switch it sometimes keeps the old duration
46
+ if not switching or random.random() > self.forget_duration_rate:
47
+ self._state["duration_minutes"] = minutes
48
+ changed = True
49
+ break
50
+
51
+ m = re.search(r"\b(\d{1,2})[:.](\d{2})\s*(am|pm)?\b", t)
52
+ if m:
53
+ hour, minute, ampm = int(m.group(1)), m.group(2), m.group(3)
54
+ if ampm == "pm" and hour < 12:
55
+ hour += 12
56
+ self._state["time"] = f"{hour:02d}:{minute}"
57
+ changed = True
58
+
59
+ for day in _DAYS:
60
+ if day in t:
61
+ self._state["day"] = day
62
+ changed = True
63
+ break
64
+
65
+ if any(w in t for w in ("yes", "confirm", "book it", "that works", "go ahead")):
66
+ self._state["confirmed"] = True
67
+ return "Booked. You will get a confirmation by text."
68
+
69
+ if changed:
70
+ bits = [str(self._state.get(k)) for k in ("service", "day", "time") if self._state.get(k)]
71
+ return "Got it, " + ", ".join(bits) + ". Shall I book that?"
72
+ return "Sorry, I did not catch that."
73
+
74
+ def outcome(self) -> Outcome:
75
+ return Outcome(entries=(), fields=dict(self._state))
@@ -0,0 +1,109 @@
1
+ """A deliberately imperfect restaurant ordering agent.
2
+
3
+ It ships so you can see real output before writing any integration, and it
4
+ fails in two ways on purpose, because an eval you cannot fail is not an eval.
5
+
6
+ 1. It drops a modifier at random under load. That is what FLAKY looks like.
7
+ 2. Its scope guard is tuned too tight, so "do you sell garlic bread" is
8
+ treated as an off topic question rather than a customer trying to order.
9
+ That is not a strawman. It is behaviour observed on a shipped agent.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import random
14
+ import re
15
+
16
+ from ..models import Entry, Outcome
17
+
18
+ MENU = {
19
+ "pepperoni pizza": 12.00,
20
+ "margherita pizza": 10.50,
21
+ "garlic bread": 4.50,
22
+ "coke": 2.00,
23
+ "sparkling water": 2.50,
24
+ }
25
+ MODIFIERS = {"thin crust", "extra cheese", "no cheese", "well done", "sliced"}
26
+ _NUM = {"one": 1, "two": 2, "three": 3, "four": 4, "five": 5, "a": 1, "an": 1}
27
+
28
+
29
+ class DemoOrderAgent:
30
+ name = "demo-order"
31
+
32
+ def __init__(self, drop_modifier_rate: float = 0.35) -> None:
33
+ self.drop_modifier_rate = drop_modifier_rate
34
+ self._lines: list[dict] = []
35
+
36
+ def reset(self) -> None:
37
+ self._lines = []
38
+
39
+ # ------------------------------------------------------------------ #
40
+
41
+ def _quantity(self, text: str) -> int:
42
+ m = re.search(r"\b(\d+)\b", text)
43
+ if m:
44
+ return int(m.group(1))
45
+ for word, n in _NUM.items():
46
+ if re.search(rf"\b{word}\b", text):
47
+ return n
48
+ return 1
49
+
50
+ def _find_item(self, text: str) -> str | None:
51
+ for item in sorted(MENU, key=len, reverse=True):
52
+ if item in text:
53
+ return item
54
+ if "pizza" in text:
55
+ return "pepperoni pizza"
56
+ return None
57
+
58
+ def send(self, text: str) -> str:
59
+ t = text.lower().strip()
60
+
61
+ # the scope guard, deliberately too tight
62
+ if t.startswith(("do you", "have you got", "is there")) and "?" in text:
63
+ return "I can only take orders from the menu, I am afraid."
64
+
65
+ if any(w in t for w in ("cancel", "remove", "take off", "drop the")):
66
+ target = self._find_item(t)
67
+ for i in range(len(self._lines) - 1, -1, -1):
68
+ if target is None or self._lines[i]["name"] == target:
69
+ removed = self._lines.pop(i)
70
+ return f"Removed the {removed['name']}."
71
+ return "There is nothing like that on your order."
72
+
73
+ mods = [m for m in MODIFIERS if m in t]
74
+ item = self._find_item(t)
75
+
76
+ # modifiers with no item named apply to the most recent line
77
+ if item is None and mods and self._lines:
78
+ line = self._lines[-1]
79
+ for m in mods:
80
+ if random.random() > self.drop_modifier_rate:
81
+ if m not in line["tags"]:
82
+ line["tags"].append(m)
83
+ return f"Noted on the {line['name']}."
84
+
85
+ if item is None:
86
+ if re.search(r"\bmake (it|that|those|them)\b", t):
87
+ if self._lines:
88
+ self._lines[-1]["quantity"] = self._quantity(t)
89
+ return f"Updated to {self._lines[-1]['quantity']}."
90
+ return "Sorry, I did not catch an item."
91
+
92
+ qty = self._quantity(t)
93
+ kept = [m for m in mods if random.random() > self.drop_modifier_rate]
94
+ self._lines.append({"name": item, "quantity": qty, "unit_price": MENU[item], "tags": kept})
95
+ return f"Added {qty} {item}."
96
+
97
+ def outcome(self) -> Outcome:
98
+ total = round(sum(l["quantity"] * l["unit_price"] for l in self._lines), 2)
99
+ return Outcome(
100
+ entries=tuple(
101
+ Entry(
102
+ name=l["name"],
103
+ numbers={"quantity": float(l["quantity"]), "unit_price": float(l["unit_price"])},
104
+ tags=tuple(l["tags"]),
105
+ )
106
+ for l in self._lines
107
+ ),
108
+ fields={"total": total},
109
+ )
@@ -0,0 +1,175 @@
1
+ """LiveKit Agents driver.
2
+
3
+ Not imported by default. Needs `pip install outturn[livekit]`.
4
+
5
+ LiveKit ships its own roomless testing path, and this driver uses it rather
6
+ than inventing one:
7
+
8
+ async with AgentSession(llm=llm) as session:
9
+ await session.start(MyAgent())
10
+ result = await session.run(user_input="two pepperoni pizzas")
11
+
12
+ No room, no audio, no telephony. Fast enough to run in CI on every commit.
13
+
14
+ THE ONE DESIGN DECISION WORTH KNOWING
15
+
16
+ The outcome does not live in LiveKit. It lives in whatever your own function
17
+ tools wrote it to, usually the session userdata, and only you know its shape.
18
+ So this driver does the session work and you pass one callable that reads
19
+ your state and returns an Outcome. The driver stays thin and never breaks
20
+ because someone structured their state differently.
21
+
22
+ USAGE
23
+
24
+ from dataclasses import dataclass, field
25
+ from livekit.agents import AgentSession, inference
26
+ from outturn.adapters.livekit import LiveKitAgent
27
+ from outturn.models import Outcome
28
+
29
+ @dataclass
30
+ class Booking:
31
+ service: str | None = None
32
+ day: str | None = None
33
+ confirmed: bool = False
34
+
35
+ async def make_session():
36
+ llm = inference.LLM(model="openai/gpt-4o-mini")
37
+ state = Booking()
38
+ session = AgentSession(llm=llm, userdata=state)
39
+ return session, MyAgent(), state
40
+
41
+ def read_outcome(state: Booking) -> Outcome:
42
+ return Outcome(fields={
43
+ "service": state.service,
44
+ "day": state.day,
45
+ "confirmed": state.confirmed,
46
+ })
47
+
48
+ agent = LiveKitAgent(make_session, read_outcome)
49
+
50
+ # then: outturn scenarios/ --adapter myproject.evals:agent
51
+
52
+ WHAT THIS DOES NOT TEST
53
+
54
+ Speech to text, endpointing, turn taking. It measures whether your agent
55
+ understands correctly, not whether it hears correctly. Both matter.
56
+ """
57
+ from __future__ import annotations
58
+
59
+ import asyncio
60
+ import inspect
61
+ from typing import Any, Callable
62
+
63
+ from ..models import Outcome
64
+
65
+
66
+
67
+
68
+ class LiveKitAgent:
69
+ """Drives a LiveKit AgentSession through text turns, one scenario at a time.
70
+
71
+ session_factory async callable returning (session, agent, state).
72
+ Called fresh on every reset, so each run starts clean.
73
+ read_outcome takes that state, returns an Outcome.
74
+ timeout seconds to wait for a single turn before the run fails.
75
+ """
76
+
77
+ name = "livekit"
78
+
79
+ def __init__(
80
+ self,
81
+ session_factory: Callable[[], Any],
82
+ read_outcome: Callable[[Any], Outcome],
83
+ *,
84
+ timeout: float = 60.0,
85
+ ) -> None:
86
+ self._factory = session_factory
87
+ self._read = read_outcome
88
+ self._timeout = timeout
89
+ self._session: Any = None
90
+ self._state: Any = None
91
+ self._loop: asyncio.AbstractEventLoop | None = None
92
+
93
+ # -- event loop plumbing, because outturn's contract is sync ----------- #
94
+
95
+ def _get_loop(self) -> asyncio.AbstractEventLoop:
96
+ if self._loop is None or self._loop.is_closed():
97
+ self._loop = asyncio.new_event_loop()
98
+ asyncio.set_event_loop(self._loop)
99
+ return self._loop
100
+
101
+ def _run(self, coro):
102
+ return self._get_loop().run_until_complete(coro)
103
+
104
+ # -- the three method contract ---------------------------------------- #
105
+
106
+ def reset(self) -> None:
107
+ """Fresh session for every run. No state leaks between runs."""
108
+ self.close()
109
+
110
+ async def _start():
111
+ made = self._factory()
112
+ if inspect.isawaitable(made):
113
+ made = await made
114
+ if not isinstance(made, tuple) or len(made) != 3:
115
+ raise TypeError(
116
+ "session_factory must return (session, agent, state), "
117
+ f"got {type(made).__name__}"
118
+ )
119
+ session, agent, state = made
120
+ await session.__aenter__()
121
+ await session.start(agent)
122
+ return session, state
123
+
124
+ self._session, self._state = self._run(_start())
125
+
126
+ def send(self, text: str) -> str:
127
+ if self._session is None:
128
+ raise RuntimeError("call reset() before send()")
129
+
130
+ async def _turn() -> str:
131
+ result = await asyncio.wait_for(
132
+ self._session.run(user_input=text), timeout=self._timeout
133
+ )
134
+ return _extract_reply(result)
135
+
136
+ return self._run(_turn())
137
+
138
+ def outcome(self) -> Outcome:
139
+ if self._state is None:
140
+ raise RuntimeError("call reset() before outcome()")
141
+ return self._read(self._state)
142
+
143
+ def close(self) -> None:
144
+ if self._session is not None and self._loop and not self._loop.is_closed():
145
+ try:
146
+ self._run(self._session.__aexit__(None, None, None))
147
+ except Exception:
148
+ pass
149
+ self._session = None
150
+ self._state = None
151
+
152
+
153
+ def _extract_reply(result: Any) -> str:
154
+ """Extract the assistant's text from a RunResult.
155
+
156
+ Verified against livekit-agents 1.8.2: the reply is a ChatMessageEvent
157
+ in result.events with item.role == 'assistant' and item.content as a list
158
+ of strings. Returns the last such message, or "" if none is found.
159
+
160
+ Returns "" rather than raising so the outcome assertion can still run.
161
+ expect_reply_contains is meant to be used sparingly anyway.
162
+ """
163
+ events = getattr(result, "events", None) or []
164
+ for event in reversed(list(events)):
165
+ item = getattr(event, "item", None)
166
+ if item is None:
167
+ continue
168
+ if getattr(item, "role", None) != "assistant":
169
+ continue
170
+ content = getattr(item, "content", None)
171
+ if isinstance(content, list) and content:
172
+ first = content[0]
173
+ if isinstance(first, str) and first:
174
+ return first
175
+ return ""
outturn/assertions.py ADDED
@@ -0,0 +1,176 @@
1
+ """The diff engine.
2
+
3
+ One rule governs every line here: numbers are compared exactly, words are
4
+ compared fuzzily. A quantity of 3 is not "close to" 2 and a duration of 90
5
+ minutes is not "close to" 60. But "deep tissue massage" and "Deep Tissue
6
+ Massage (60min)" are the same service.
7
+
8
+ Everywhere being clever would have been possible, being clever would have
9
+ hidden a real failure behind a generous match. That is the one thing a test
10
+ harness must never do.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import re
15
+ from difflib import SequenceMatcher
16
+ from typing import Any
17
+
18
+ from .models import Discrepancy, Entry, Kind, Outcome
19
+
20
+ _PUNCT = re.compile(r"[^a-z0-9 ]+")
21
+ _SPACE = re.compile(r"\s+")
22
+ # words carrying no identity, dropped before comparison
23
+ _NOISE = {"a", "an", "the", "of", "with", "and", "please", "order", "for"}
24
+
25
+ DEFAULT_NAME_THRESHOLD = 0.82
26
+ DEFAULT_TAG_THRESHOLD = 0.80
27
+ DEFAULT_TOLERANCE = 0.005
28
+
29
+
30
+ def normalize(text: str) -> str:
31
+ """Lowercase, strip punctuation, drop noise words, crudely singularize.
32
+
33
+ Deliberately simple. If you need a stemmer here, your names are the
34
+ problem, not the matcher.
35
+ """
36
+ t = _PUNCT.sub(" ", str(text).lower())
37
+ words = [w for w in _SPACE.sub(" ", t).strip().split(" ") if w and w not in _NOISE]
38
+ out = []
39
+ for w in words:
40
+ if len(w) > 3 and w.endswith("es") and not w.endswith("ses"):
41
+ w = w[:-2]
42
+ elif len(w) > 3 and w.endswith("s") and not w.endswith("ss"):
43
+ w = w[:-1]
44
+ out.append(w)
45
+ return " ".join(sorted(out))
46
+
47
+
48
+ def similarity(a: str, b: str) -> float:
49
+ return SequenceMatcher(None, normalize(a), normalize(b)).ratio()
50
+
51
+
52
+ def _numbers_equal(a: float, b: float, tolerance: float) -> bool:
53
+ return abs(float(a) - float(b)) <= tolerance
54
+
55
+
56
+ def _number_agreement(want: Entry, have: Entry, tolerance: float) -> int:
57
+ """How many of the expected numbers this candidate also matches.
58
+
59
+ Used only to break ties between candidates whose names score identically.
60
+ Two cart lines called "coffee" are indistinguishable by name, so the one
61
+ whose quantity also matches is the intended pairing. vaevals picked
62
+ whichever came first, which made a real bug look like two.
63
+ """
64
+ return sum(
65
+ 1
66
+ for k, v in want.numbers.items()
67
+ if k in have.numbers and _numbers_equal(v, have.numbers[k], tolerance)
68
+ )
69
+
70
+
71
+ def _match_tags(
72
+ expected: tuple[str, ...],
73
+ actual: tuple[str, ...],
74
+ where: str,
75
+ threshold: float,
76
+ ) -> list[Discrepancy]:
77
+ out: list[Discrepancy] = []
78
+ remaining = list(actual)
79
+ for want in expected:
80
+ best_i, best_score = -1, 0.0
81
+ for i, have in enumerate(remaining):
82
+ s = similarity(want, have)
83
+ if s > best_score:
84
+ best_i, best_score = i, s
85
+ if best_score >= threshold:
86
+ remaining.pop(best_i)
87
+ else:
88
+ out.append(Discrepancy(Kind.MISSING_TAG, where, expected=want, actual=None))
89
+ for leftover in remaining:
90
+ out.append(Discrepancy(Kind.EXTRA_TAG, where, expected=None, actual=leftover))
91
+ return out
92
+
93
+
94
+ def _compare_fields(
95
+ expected: dict[str, Any],
96
+ actual: dict[str, Any],
97
+ *,
98
+ tag_threshold: float,
99
+ tolerance: float,
100
+ ) -> list[Discrepancy]:
101
+ """Scalars. Numbers and booleans exact, strings fuzzy.
102
+
103
+ Extra fields in the actual outcome are ignored on purpose. The scenario
104
+ states what must be true, not everything that may be present.
105
+ """
106
+ out: list[Discrepancy] = []
107
+ for key, want in expected.items():
108
+ if key not in actual or actual[key] is None:
109
+ out.append(Discrepancy(Kind.MISSING_FIELD, key, expected=want, actual=None))
110
+ continue
111
+ have = actual[key]
112
+
113
+ if isinstance(want, bool) or isinstance(have, bool):
114
+ if bool(want) != bool(have):
115
+ out.append(Discrepancy(Kind.WRONG_FIELD, key, expected=want, actual=have))
116
+ elif isinstance(want, (int, float)) and isinstance(have, (int, float)):
117
+ if not _numbers_equal(want, have, tolerance):
118
+ out.append(Discrepancy(Kind.WRONG_FIELD, key, expected=want, actual=have))
119
+ else:
120
+ if similarity(str(want), str(have)) < tag_threshold:
121
+ out.append(Discrepancy(Kind.WRONG_FIELD, key, expected=want, actual=have))
122
+ return out
123
+
124
+
125
+ def compare_outcomes(
126
+ expected: Outcome,
127
+ actual: Outcome,
128
+ *,
129
+ name_threshold: float = DEFAULT_NAME_THRESHOLD,
130
+ tag_threshold: float = DEFAULT_TAG_THRESHOLD,
131
+ tolerance: float = DEFAULT_TOLERANCE,
132
+ ) -> list[Discrepancy]:
133
+ """Best-match on entry names, exact on every number, fuzzy on every word."""
134
+ out: list[Discrepancy] = []
135
+ unmatched: list[Entry] = list(actual.entries)
136
+
137
+ for want in expected.entries:
138
+ best_i, best_key = -1, (0.0, -1)
139
+ for i, have in enumerate(unmatched):
140
+ score = similarity(want.name, have.name)
141
+ key = (score, _number_agreement(want, have, tolerance))
142
+ if key > best_key:
143
+ best_i, best_key = i, key
144
+
145
+ if best_key[0] < name_threshold:
146
+ out.append(
147
+ Discrepancy(Kind.MISSING_ENTRY, "not found", expected=want.name, actual=None)
148
+ )
149
+ continue
150
+
151
+ have = unmatched.pop(best_i)
152
+
153
+ for key, wanted in want.numbers.items():
154
+ if key not in have.numbers:
155
+ out.append(
156
+ Discrepancy(Kind.MISSING_NUMBER, f"{key} on {want.name!r}",
157
+ expected=wanted, actual=None)
158
+ )
159
+ elif not _numbers_equal(wanted, have.numbers[key], tolerance):
160
+ out.append(
161
+ Discrepancy(Kind.WRONG_NUMBER, f"{key} on {want.name!r}",
162
+ expected=wanted, actual=have.numbers[key])
163
+ )
164
+
165
+ out.extend(_match_tags(want.tags, have.tags, f"on {want.name!r}", tag_threshold))
166
+
167
+ for leftover in unmatched:
168
+ out.append(
169
+ Discrepancy(Kind.EXTRA_ENTRY, "not expected", expected=None, actual=leftover.name)
170
+ )
171
+
172
+ out.extend(
173
+ _compare_fields(expected.fields, actual.fields,
174
+ tag_threshold=tag_threshold, tolerance=tolerance)
175
+ )
176
+ return out
outturn/cli.py ADDED
@@ -0,0 +1,58 @@
1
+ """Command line entry point."""
2
+ from __future__ import annotations
3
+
4
+ import argparse
5
+ import random
6
+ import sys
7
+ from pathlib import Path
8
+
9
+ from . import adapters, report, runner, scenario
10
+
11
+
12
+ def build_parser() -> argparse.ArgumentParser:
13
+ p = argparse.ArgumentParser(
14
+ prog="outturn",
15
+ description="Did the agent produce the right outcome? As a number.",
16
+ )
17
+ p.add_argument("path", type=Path, help="a scenario file, or a directory of them")
18
+ p.add_argument("--adapter", default="demo-order",
19
+ help="a built in name, or an import path like myproject.agents:MyAgent")
20
+ p.add_argument("--runs", type=int, default=5, help="runs per scenario (default 5)")
21
+ p.add_argument("--seed", type=int, default=None, help="seed the RNG for a repeatable run")
22
+ p.add_argument("--threshold", type=float, default=None,
23
+ help="exit 1 if overall accuracy falls below this (0 to 1)")
24
+ p.add_argument("--json", action="store_true", help="machine readable output")
25
+ p.add_argument("--list-adapters", action="store_true", help="show registered adapters and exit")
26
+ return p
27
+
28
+
29
+ def main(argv: list[str] | None = None) -> int:
30
+ args = build_parser().parse_args(argv)
31
+
32
+ if args.list_adapters:
33
+ for name in sorted(adapters.BUILTIN):
34
+ print(name)
35
+ print("\nor any import path, for example myproject.agents:MyAgent")
36
+ return 0
37
+
38
+ if args.runs < 1:
39
+ print("--runs must be at least 1", file=sys.stderr)
40
+ return 2
41
+ if args.seed is not None:
42
+ random.seed(args.seed)
43
+
44
+ try:
45
+ scenarios = scenario.load(args.path)
46
+ agent = adapters.get(args.adapter)
47
+ except (scenario.ScenarioError, adapters.AdapterError, OSError) as e:
48
+ print(str(e).strip("'"), file=sys.stderr)
49
+ return 2
50
+
51
+ reports = runner.run_all(agent, scenarios, args.runs)
52
+ print(report.as_json(reports) if args.json else report.render(reports))
53
+
54
+ if args.threshold is not None:
55
+ passes, runs = report.overall(reports)
56
+ if runs and (passes / runs) < args.threshold:
57
+ return 1
58
+ return 0
outturn/models.py ADDED
@@ -0,0 +1,158 @@
1
+ """Core data types.
2
+
3
+ One idea runs through all of it: an agent conversation ends in a structured
4
+ outcome, and a structured outcome can be compared exactly. A restaurant order
5
+ is one kind of outcome. So is a booking, a triage, a qualified lead.
6
+
7
+ Two comparison rules, applied everywhere:
8
+ numbers are compared exactly a quantity of 3 is not close to 2
9
+ words are compared fuzzily "deep tissue" and "Deep Tissue Massage" match
10
+ """
11
+ from __future__ import annotations
12
+
13
+ from dataclasses import dataclass, field
14
+ from enum import Enum
15
+ from typing import Any
16
+
17
+
18
+ @dataclass(frozen=True)
19
+ class Entry:
20
+ """One repeated thing in an outcome.
21
+
22
+ A cart line. A passenger. A prescription. Anything the agent can produce
23
+ more than one of.
24
+
25
+ name matched fuzzily, it is how the entry is identified
26
+ numbers matched exactly, quantity, price, duration, dosage
27
+ tags matched fuzzily as a set, modifiers, options, flags
28
+ """
29
+
30
+ name: str
31
+ numbers: dict[str, float] = field(default_factory=dict)
32
+ tags: tuple[str, ...] = ()
33
+
34
+ @staticmethod
35
+ def from_dict(d: dict[str, Any]) -> "Entry":
36
+ if "name" not in d:
37
+ raise ValueError("an entry needs a 'name'")
38
+ numbers = {
39
+ str(k): float(v)
40
+ for k, v in d.items()
41
+ if k not in ("name", "tags") and isinstance(v, (int, float)) and not isinstance(v, bool)
42
+ }
43
+ tags = d.get("tags", ())
44
+ if isinstance(tags, str):
45
+ tags = (tags,)
46
+ return Entry(name=str(d["name"]), numbers=numbers, tags=tuple(str(t) for t in tags))
47
+
48
+
49
+ @dataclass(frozen=True)
50
+ class Outcome:
51
+ """What the conversation actually produced.
52
+
53
+ entries the repeated things, may be empty
54
+ fields the scalars. numbers and booleans exact, strings fuzzy
55
+
56
+ A restaurant order is entries plus a total field.
57
+ A booking is no entries at all, just fields.
58
+ Both work.
59
+ """
60
+
61
+ entries: tuple[Entry, ...] = ()
62
+ fields: dict[str, Any] = field(default_factory=dict)
63
+
64
+ @staticmethod
65
+ def from_dict(d: dict[str, Any] | None) -> "Outcome":
66
+ d = dict(d or {})
67
+ raw_entries = d.pop("entries", ())
68
+ return Outcome(
69
+ entries=tuple(Entry.from_dict(e) for e in raw_entries),
70
+ fields={str(k): v for k, v in d.items()},
71
+ )
72
+
73
+ def is_empty(self) -> bool:
74
+ return not self.entries and not self.fields
75
+
76
+
77
+ @dataclass(frozen=True)
78
+ class Turn:
79
+ """One thing the caller says."""
80
+
81
+ user: str
82
+ # substrings the reply must contain. use sparingly: asserting on phrasing
83
+ # is brittle by nature, which is the problem this tool exists to avoid.
84
+ expect_reply_contains: tuple[str, ...] = ()
85
+
86
+
87
+ @dataclass(frozen=True)
88
+ class Scenario:
89
+ id: str
90
+ turns: tuple[Turn, ...]
91
+ expect: Outcome
92
+ description: str = ""
93
+ tags: tuple[str, ...] = ()
94
+ runs: int | None = None # overrides the global run count
95
+
96
+
97
+ class Kind(str, Enum):
98
+ MISSING_ENTRY = "missing_entry"
99
+ EXTRA_ENTRY = "extra_entry"
100
+ WRONG_NUMBER = "wrong_number"
101
+ MISSING_NUMBER = "missing_number"
102
+ MISSING_TAG = "missing_tag"
103
+ EXTRA_TAG = "extra_tag"
104
+ MISSING_FIELD = "missing_field"
105
+ WRONG_FIELD = "wrong_field"
106
+ REPLY_MISSING_TEXT = "reply_missing_text"
107
+
108
+
109
+ @dataclass(frozen=True)
110
+ class Discrepancy:
111
+ kind: Kind
112
+ detail: str
113
+ expected: Any = None
114
+ actual: Any = None
115
+
116
+ def __str__(self) -> str:
117
+ if self.expected is None and self.actual is None:
118
+ return f"{self.kind.value}: {self.detail}"
119
+ return f"{self.kind.value}: {self.detail} (expected {self.expected!r}, got {self.actual!r})"
120
+
121
+
122
+ @dataclass
123
+ class RunResult:
124
+ scenario_id: str
125
+ run_index: int
126
+ discrepancies: list[Discrepancy] = field(default_factory=list)
127
+ transcript: list[tuple[str, str]] = field(default_factory=list)
128
+ error: str | None = None
129
+
130
+ @property
131
+ def passed(self) -> bool:
132
+ return self.error is None and not self.discrepancies
133
+
134
+
135
+ @dataclass
136
+ class ScenarioReport:
137
+ scenario_id: str
138
+ results: list[RunResult] = field(default_factory=list)
139
+
140
+ @property
141
+ def runs(self) -> int:
142
+ return len(self.results)
143
+
144
+ @property
145
+ def passes(self) -> int:
146
+ return sum(1 for r in self.results if r.passed)
147
+
148
+ @property
149
+ def pass_rate(self) -> float:
150
+ return (self.passes / self.runs) if self.runs else 0.0
151
+
152
+ @property
153
+ def is_flaky(self) -> bool:
154
+ """Passed sometimes and failed sometimes.
155
+
156
+ Worse than always failing, because it ships.
157
+ """
158
+ return 0 < self.passes < self.runs
outturn/report.py ADDED
@@ -0,0 +1,72 @@
1
+ """Turns reports into something a person reads in five seconds."""
2
+ from __future__ import annotations
3
+
4
+ import json
5
+ from collections import Counter
6
+
7
+ from .models import ScenarioReport
8
+
9
+
10
+ def _failure_lines(rep: ScenarioReport, limit: int = 3) -> list[str]:
11
+ counts: Counter[str] = Counter()
12
+ for r in rep.results:
13
+ if r.error:
14
+ counts[f"error: {r.error}"] += 1
15
+ for d in r.discrepancies:
16
+ counts[str(d)] += 1
17
+ return [f" {n}x {text}" for text, n in counts.most_common(limit)]
18
+
19
+
20
+ def overall(reports: list[ScenarioReport]) -> tuple[int, int]:
21
+ passes = sum(r.passes for r in reports)
22
+ runs = sum(r.runs for r in reports)
23
+ return passes, runs
24
+
25
+
26
+ def render(reports: list[ScenarioReport]) -> str:
27
+ lines: list[str] = []
28
+ for rep in reports:
29
+ flag = " FLAKY" if rep.is_flaky else ""
30
+ lines.append(
31
+ f"{rep.pass_rate * 100:5.0f}% {rep.scenario_id:<34} {rep.passes}/{rep.runs}{flag}"
32
+ )
33
+ if rep.passes < rep.runs:
34
+ lines.extend(_failure_lines(rep))
35
+
36
+ passes, runs = overall(reports)
37
+ rate = (passes / runs * 100) if runs else 0.0
38
+ lines.append("")
39
+ lines.append(f"outcome accuracy: {rate:.1f}% ({passes}/{runs} runs)")
40
+
41
+ flaky = [r.scenario_id for r in reports if r.is_flaky]
42
+ if flaky:
43
+ lines.append(f"flaky scenarios ({len(flaky)}): {', '.join(flaky)}")
44
+ lines.append("a scenario that passes sometimes is a bug that ships sometimes.")
45
+ return "\n".join(lines)
46
+
47
+
48
+ def as_json(reports: list[ScenarioReport]) -> str:
49
+ passes, runs = overall(reports)
50
+ return json.dumps(
51
+ {
52
+ "accuracy": (passes / runs) if runs else 0.0,
53
+ "passes": passes,
54
+ "runs": runs,
55
+ "flaky": [r.scenario_id for r in reports if r.is_flaky],
56
+ "scenarios": [
57
+ {
58
+ "id": r.scenario_id,
59
+ "passes": r.passes,
60
+ "runs": r.runs,
61
+ "pass_rate": r.pass_rate,
62
+ "flaky": r.is_flaky,
63
+ "failures": sorted(
64
+ {str(d) for res in r.results for d in res.discrepancies}
65
+ ),
66
+ "errors": sorted({res.error for res in r.results if res.error}),
67
+ }
68
+ for r in reports
69
+ ],
70
+ },
71
+ indent=2,
72
+ )
outturn/runner.py ADDED
@@ -0,0 +1,42 @@
1
+ """Runs each scenario N times.
2
+
3
+ N, not once. Agents are nondeterministic, so one green run is not evidence.
4
+ A bug that appears one run in five is invisible to a boolean test and obvious
5
+ in a percentage.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ from .adapters.base import Agent
10
+ from .assertions import compare_outcomes
11
+ from .models import Discrepancy, Kind, RunResult, Scenario, ScenarioReport
12
+
13
+
14
+ def run_once(agent: Agent, scenario: Scenario, run_index: int) -> RunResult:
15
+ result = RunResult(scenario_id=scenario.id, run_index=run_index)
16
+ try:
17
+ agent.reset()
18
+ for turn in scenario.turns:
19
+ reply = agent.send(turn.user)
20
+ result.transcript.append((turn.user, reply))
21
+ for needle in turn.expect_reply_contains:
22
+ if needle.lower() not in (reply or "").lower():
23
+ result.discrepancies.append(
24
+ Discrepancy(Kind.REPLY_MISSING_TEXT, f"after {turn.user!r}",
25
+ expected=needle, actual=reply)
26
+ )
27
+ result.discrepancies.extend(compare_outcomes(scenario.expect, agent.outcome()))
28
+ except Exception as e: # an agent that crashes is a failing run, not a crashing suite
29
+ result.error = f"{type(e).__name__}: {e}"
30
+ return result
31
+
32
+
33
+ def run_scenario(agent: Agent, scenario: Scenario, runs: int) -> ScenarioReport:
34
+ n = scenario.runs if scenario.runs is not None else runs
35
+ report = ScenarioReport(scenario_id=scenario.id)
36
+ for i in range(n):
37
+ report.results.append(run_once(agent, scenario, i))
38
+ return report
39
+
40
+
41
+ def run_all(agent: Agent, scenarios: list[Scenario], runs: int) -> list[ScenarioReport]:
42
+ return [run_scenario(agent, s, runs) for s in scenarios]
outturn/scenario.py ADDED
@@ -0,0 +1,67 @@
1
+ """Load scenarios from YAML. Fail loudly on a bad file, never guess."""
2
+ from __future__ import annotations
3
+
4
+ from pathlib import Path
5
+
6
+ import yaml
7
+
8
+ from .models import Outcome, Scenario, Turn
9
+
10
+
11
+ class ScenarioError(ValueError):
12
+ pass
13
+
14
+
15
+ def _turn(raw: object, ctx: str) -> Turn:
16
+ if isinstance(raw, str):
17
+ return Turn(user=raw)
18
+ if isinstance(raw, dict):
19
+ if "user" not in raw:
20
+ raise ScenarioError(f"{ctx}: a turn needs a 'user' key")
21
+ contains = raw.get("expect_reply_contains", ())
22
+ if isinstance(contains, str):
23
+ contains = (contains,)
24
+ return Turn(user=str(raw["user"]), expect_reply_contains=tuple(str(c) for c in contains))
25
+ raise ScenarioError(f"{ctx}: a turn must be a string or a mapping, got {type(raw).__name__}")
26
+
27
+
28
+ def load_file(path: Path) -> Scenario:
29
+ raw = yaml.safe_load(path.read_text())
30
+ if not isinstance(raw, dict):
31
+ raise ScenarioError(f"{path}: top level must be a mapping")
32
+ for key in ("id", "turns", "expect"):
33
+ if key not in raw:
34
+ raise ScenarioError(f"{path}: missing required key {key!r}")
35
+ turns = raw["turns"]
36
+ if not isinstance(turns, list) or not turns:
37
+ raise ScenarioError(f"{path}: 'turns' must be a non empty list")
38
+ try:
39
+ expect = Outcome.from_dict(raw["expect"])
40
+ except ValueError as e:
41
+ raise ScenarioError(f"{path}: bad 'expect' block, {e}") from e
42
+ if expect.is_empty():
43
+ raise ScenarioError(f"{path}: 'expect' asserts nothing. A scenario that cannot fail is not a test.")
44
+ return Scenario(
45
+ id=str(raw["id"]),
46
+ description=str(raw.get("description", "")),
47
+ turns=tuple(_turn(t, str(path)) for t in turns),
48
+ expect=expect,
49
+ tags=tuple(str(t) for t in raw.get("tags", ())),
50
+ runs=(int(raw["runs"]) if raw.get("runs") is not None else None),
51
+ )
52
+
53
+
54
+ def load(target: Path) -> list[Scenario]:
55
+ """A single file, or every .yaml under a directory, recursively."""
56
+ if target.is_file():
57
+ return [load_file(target)]
58
+ files = sorted(p for p in target.rglob("*.y*ml") if p.is_file())
59
+ if not files:
60
+ raise ScenarioError(f"{target}: no .yaml scenario files found")
61
+ scenarios = [load_file(p) for p in files]
62
+ seen: set[str] = set()
63
+ for s in scenarios:
64
+ if s.id in seen:
65
+ raise ScenarioError(f"duplicate scenario id {s.id!r}")
66
+ seen.add(s.id)
67
+ return scenarios
@@ -0,0 +1,240 @@
1
+ Metadata-Version: 2.4
2
+ Name: outturn
3
+ Version: 0.1.0
4
+ Summary: Did the agent produce the right outcome? As a number.
5
+ Author: Wisdom Omons
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/OsasDTEch/outturn
8
+ Project-URL: Source, https://github.com/OsasDTEch/outturn
9
+ Keywords: voice-agents,llm,evaluation,testing,livekit,agents
10
+ Classifier: Development Status :: 3 - Alpha
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Topic :: Software Development :: Testing
14
+ Requires-Python: >=3.10
15
+ Description-Content-Type: text/markdown
16
+ License-File: LICENSE
17
+ Requires-Dist: PyYAML>=6.0
18
+ Provides-Extra: dev
19
+ Requires-Dist: pytest>=7.4; extra == "dev"
20
+ Provides-Extra: livekit
21
+ Requires-Dist: livekit-agents==1.8.2; extra == "livekit"
22
+ Dynamic: license-file
23
+
24
+ # outturn
25
+
26
+ **Did the agent produce the right outcome? As a number.**
27
+
28
+ A conversation does not end in words. It ends in an order, a booking, a routed ticket, a qualified lead. That is structured data, and structured data can be compared exactly. So whether your agent got it right does not have to be something you learn from a refund. It can be a percentage that runs on every commit.
29
+
30
+ ```
31
+ 100% simple_booking 8/8
32
+ 50% service_switch_duration 4/8 FLAKY
33
+ 4x wrong_field: duration_minutes (expected 90, got 60)
34
+ 100% never_confirmed 8/8
35
+
36
+ outcome accuracy: 83.3% (20/24 runs)
37
+ flaky scenarios (1): service_switch_duration
38
+ a scenario that passes sometimes is a bug that ships sometimes.
39
+ ```
40
+
41
+ ---
42
+
43
+ ## Why this exists
44
+
45
+ Most voice agent testing is a person calling the number and listening. That finds the bug in front of you and none of the others, it cannot run in CI, and it produces an opinion instead of a number.
46
+
47
+ The alternatives are not much better. Asserting on the transcript is asserting on phrasing, which changes every time you touch the prompt. Using a model to grade the conversation means one nondeterministic system judging another.
48
+
49
+ The outcome is the way out. It is the thing the customer actually receives, it is already structured, and it can be compared exactly.
50
+
51
+ ## Three design decisions
52
+
53
+ **Numbers are exact, words are fuzzy.** A quantity of 3 is not close to 2, and a 90 minute appointment is not close to a 60 minute one. But "deep tissue massage" and "Deep Tissue Massage (60min)" are the same service. So every number is compared exactly and every word by normalized similarity.
54
+
55
+ **Every scenario runs N times and reports a pass rate.** Agents are nondeterministic. One green run is not evidence. A bug that appears one run in five is invisible to a boolean test and obvious in a percentage, so outturn reports rates and flags anything that passed sometimes and failed sometimes as FLAKY. Intermittent is worse than broken, because intermittent hides.
56
+
57
+ **An outcome is repeated things plus scalars.** That is all. An order is entries plus a total. A booking is no entries at all, just fields. A triage is fields. One model, no special cases per domain, which is why the same engine handles an ordering agent and a scheduling agent without a line of new code.
58
+
59
+ ## Install
60
+
61
+ ```bash
62
+ pip install outturn
63
+ ```
64
+
65
+ Or from source, if you want the demo scenarios to play with:
66
+
67
+ ```bash
68
+ git clone https://github.com/OsasDTEch/outturn
69
+ cd outturn
70
+ pip install -e ".[dev]"
71
+ ```
72
+
73
+ ## Run it right now
74
+
75
+ Two deliberately imperfect demo agents ship with the repo, so you can see real output before writing any integration.
76
+
77
+ ```bash
78
+ outturn scenarios/order --adapter demo-order --runs 8 --seed 42
79
+ outturn scenarios/booking --adapter demo-booking --runs 8 --seed 42
80
+ ```
81
+
82
+ Each fails on purpose, because an eval you cannot fail is not an eval.
83
+
84
+ The ordering agent drops a modifier at random under load, which is what FLAKY looks like, and its scope guard is tuned too tight so "do you sell garlic bread" is treated as an off topic question rather than a customer trying to order. That second one is not a strawman, it is behaviour observed on a shipped restaurant agent.
85
+
86
+ The booking agent forgets to carry the duration across when the customer switches service mid call. A 90 minute massage in a 60 minute slot double books the therapist, and nobody finds out until the day.
87
+
88
+ ## Writing a scenario
89
+
90
+ Turns in, expected outcome out.
91
+
92
+ **An order**, which has line items:
93
+
94
+ ```yaml
95
+ id: modifier_stacking
96
+ description: >
97
+ Two modifiers on one item, added in a separate turn from the item itself.
98
+ tags: [order, modifiers]
99
+
100
+ turns:
101
+ - "Hi, can I get two large pepperoni pizzas"
102
+ - "Actually make those thin crust"
103
+ - "And extra cheese on them please"
104
+
105
+ expect:
106
+ entries:
107
+ - name: pepperoni pizza
108
+ quantity: 2
109
+ unit_price: 12.00
110
+ tags: [thin crust, extra cheese]
111
+ total: 24.00
112
+ ```
113
+
114
+ **A booking**, which has none:
115
+
116
+ ```yaml
117
+ id: service_switch_duration
118
+ turns:
119
+ - "Can I book a deep tissue massage for Friday"
120
+ - "Actually make it a sports massage instead"
121
+ - "4:00 pm, and yes please book it"
122
+
123
+ expect:
124
+ service: sports massage
125
+ day: friday
126
+ time: "16:00"
127
+ duration_minutes: 90
128
+ confirmed: true
129
+ ```
130
+
131
+ Same engine. Anything numeric under an entry becomes an exact comparison. Anything at the top level is a field: numbers and booleans exact, strings fuzzy.
132
+
133
+ You can also assert on a reply, though use it sparingly since phrasing is the brittle part:
134
+
135
+ ```yaml
136
+ turns:
137
+ - user: "Do you sell garlic bread?"
138
+ expect_reply_contains: ["garlic bread"]
139
+ ```
140
+
141
+ ## Connecting your own agent
142
+
143
+ Implement three methods.
144
+
145
+ ```python
146
+ from outturn.models import Outcome, Entry
147
+
148
+ class MyAgent:
149
+ name = "my-agent"
150
+
151
+ def reset(self) -> None:
152
+ """Fresh conversation. Called before every run."""
153
+ self.session = start_session()
154
+
155
+ def send(self, text: str) -> str:
156
+ """One caller turn in, the agent's reply out."""
157
+ return self.session.turn(text)
158
+
159
+ def outcome(self) -> Outcome:
160
+ """The structured result, right now."""
161
+ return Outcome(
162
+ entries=tuple(
163
+ Entry(name=l.name,
164
+ numbers={"quantity": l.qty, "unit_price": l.price},
165
+ tags=tuple(l.modifiers))
166
+ for l in self.session.order.lines
167
+ ),
168
+ fields={"total": self.session.order.total},
169
+ )
170
+ ```
171
+
172
+ Then point outturn at it. **Your agent lives in your repo, not in this one.**
173
+
174
+ ```bash
175
+ outturn scenarios/ --adapter myproject.agents:MyAgent --runs 10
176
+ ```
177
+
178
+ Anything importable works. If the object is missing one of the three methods,
179
+ outturn says so before the run starts rather than failing with an
180
+ AttributeError halfway through.
181
+
182
+ If your agent cannot hand back a structured outcome, outturn cannot help you, and that is itself the finding. State that lives only in the model's context is not inspectable, and what is not inspectable is not testable. Move the outcome into code and the tool works.
183
+
184
+ ## LiveKit
185
+
186
+ There is a driver at `outturn/adapters/livekit.py`. It takes a session factory and one callable that reads your state and returns an Outcome, because the outcome does not live in LiveKit, it lives in whatever your own function tools wrote it to.
187
+
188
+ **Verified against livekit-agents 1.8.2.** Tested with the ollama.com cloud API (model `gemma4:31b`, OpenAI-compatible endpoint). Confirmed:
189
+
190
+ - A session starts with no room and no audio.
191
+ - State written by function tools persists across multiple `session.run()` calls on the same session, so multi-turn scenarios work as written.
192
+ - The assistant reply is a `ChatMessageEvent` in `result.events` with `item.content[0]` as the text.
193
+
194
+ Pin your own install to `livekit-agents==1.8.2` until you have tested against a newer version. This API is young and the `RunResult` shape has changed between releases.
195
+
196
+ ## In CI
197
+
198
+ ```bash
199
+ outturn scenarios/booking --adapter demo-booking --runs 10 --threshold 0.95
200
+ ```
201
+
202
+ Exits non zero if overall accuracy falls below the threshold. Every bug you fix becomes a scenario, so it can never come back silently. `--json` gives machine readable output for tracking accuracy over time.
203
+
204
+ ## The scenarios worth writing
205
+
206
+ The ones that break agents, roughly in order of how often they do:
207
+
208
+ - **Corrections mid call.** The customer changes their mind after the agent has already recorded something. This is where most outcomes go wrong, and the failure is usually a field that did not get updated alongside the one that did.
209
+ - **Reference without naming.** "Make it three" with no item named.
210
+ - **Cancellation.** Added, then removed. Do the derived values follow?
211
+ - **Confirmation.** Did the agent act on intent rather than on an actual yes? Booking a customer who was still thinking is a real and expensive failure.
212
+ - **Name collisions.** Two items or two services that sound alike over a phone line.
213
+ - **Illegal combinations.** Extra cheese on a coke. Should be rejected by schema, not accepted politely and discovered later.
214
+ - **Out of scope questions that are really orders.** "Do you sell X" is a customer trying to buy X.
215
+
216
+ ## What this does not test
217
+
218
+ outturn drives your agent through text. That is fast enough to run in CI on every commit, and it isolates the reasoning and outcome layer from the audio layer.
219
+
220
+ It does not test speech to text, endpointing or turn taking, and those are real sources of failure, particularly on telephony where audio is narrowband and degrades worst on exactly what matters here: names, numbers and proper nouns.
221
+
222
+ **So this measures whether your agent understands correctly, not whether it hears correctly.** Both matter. For the timing half of the picture, see [voice-latency-profiler](https://github.com/OsasDTEch/voice-latency-profiler).
223
+
224
+ ## Status
225
+
226
+ v0.1. The core works and is tested, 31 tests. Roadmap, roughly in order:
227
+
228
+ - [x] LiveKit driver verified against livekit-agents 1.8.2 (ollama.com cloud, `gemma4:31b`)
229
+ - [ ] Path assertions: which tools were called, with which arguments
230
+ - [ ] Pipecat driver
231
+ - [ ] Audio mode, TTS in and STT out, for end to end runs
232
+ - [ ] Accuracy tracked over time, so regressions show as a trend
233
+
234
+ `PRD.md` and `TRD.md` in this repo cover the reasoning and the internals.
235
+
236
+ This project supersedes [voice-agent-evals](https://github.com/OsasDTEch/voice-agent-evals), which asserted on carts only. An order is one kind of outcome among many, and the narrower version was useful to about a tenth of the agents worth testing.
237
+
238
+ ## Licence
239
+
240
+ MIT. Built by [Wisdom Omons](https://linkedin.com/in/omons-wisdom).
@@ -0,0 +1,19 @@
1
+ outturn/__init__.py,sha256=7tECRjTfloJxUAFQMhiyl8Oj3DotJuN9z3lQOrWJFn8,251
2
+ outturn/__main__.py,sha256=E6Gls0DNz8GQK2K-kOUIx8cYhgANW_CH54VKrfCfs14,52
3
+ outturn/assertions.py,sha256=tTx1Sv3S1xxfKUOTZe02fIN38NQhSM_9iqVKTAzUP8E,6243
4
+ outturn/cli.py,sha256=127Hwi3too_Rz1h1DlWKXuU3YYwufZFfGz2C9D2jMOI,2115
5
+ outturn/models.py,sha256=o6a-_z8YCtE4iL5-hoAru0X2toPuft8L_3-Bow-6Ct0,4564
6
+ outturn/report.py,sha256=Ev4in-SAjtPQIPrRsLZxBydyoLhljXwyXiirpm1MeIw,2391
7
+ outturn/runner.py,sha256=RACFiJ4ZD2G0GSCpZd4FayA8pyqGIqM1iUlOB6fzxqs,1744
8
+ outturn/scenario.py,sha256=XmJ5GQ4DzUYmcTV3el4NXYwjAE9CypzFxGcxIWMICH8,2446
9
+ outturn/adapters/__init__.py,sha256=F4wvsQAcf-fgERHkQXYmXvhdqnjnBlqAiY-Ln7pty-M,2114
10
+ outturn/adapters/base.py,sha256=Xsk0ymoZbr7donfiTIzpEgq3bM7zI9-TC8trbtXgllU,755
11
+ outturn/adapters/demo_booking.py,sha256=PCkjsFxab9yoODGHU78Ua87xu0sgC6wxveANP6AXcCQ,2622
12
+ outturn/adapters/demo_order.py,sha256=w1UqGXfRzshCyBL0Vv11qYqkKmly0cxZWaDA-x11ICQ,3990
13
+ outturn/adapters/livekit.py,sha256=eA1_NTvbpzQoVVmzTfu64H9ZtfqsAglf21s4LOzG6Rg,5815
14
+ outturn-0.1.0.dist-info/licenses/LICENSE,sha256=kM3jvrV9E_rQdz7LH8CfBIuunYVt5FIXd2kiVoRQu08,1069
15
+ outturn-0.1.0.dist-info/METADATA,sha256=D1a-Jhq33C_MhXE9pOVEzvXKwE7fgj-aRzYyU_zVWqU,10673
16
+ outturn-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
17
+ outturn-0.1.0.dist-info/entry_points.txt,sha256=Q1mlhUgk2gC4AlyKx2A2JJGi5nuZd5uZctaGgM1p2E8,45
18
+ outturn-0.1.0.dist-info/top_level.txt,sha256=2kGQyWMtSL4ys27HOT8DpQ28PZg_VqKC-W6s96hz9bo,8
19
+ outturn-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ outturn = outturn.cli:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Wisdom Omons
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ outturn