outturn 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- outturn/__init__.py +5 -0
- outturn/__main__.py +5 -0
- outturn/adapters/__init__.py +68 -0
- outturn/adapters/base.py +25 -0
- outturn/adapters/demo_booking.py +75 -0
- outturn/adapters/demo_order.py +109 -0
- outturn/adapters/livekit.py +175 -0
- outturn/assertions.py +176 -0
- outturn/cli.py +58 -0
- outturn/models.py +158 -0
- outturn/report.py +72 -0
- outturn/runner.py +42 -0
- outturn/scenario.py +67 -0
- outturn-0.1.0.dist-info/METADATA +240 -0
- outturn-0.1.0.dist-info/RECORD +19 -0
- outturn-0.1.0.dist-info/WHEEL +5 -0
- outturn-0.1.0.dist-info/entry_points.txt +2 -0
- outturn-0.1.0.dist-info/licenses/LICENSE +21 -0
- outturn-0.1.0.dist-info/top_level.txt +1 -0
outturn/__init__.py
ADDED
outturn/__main__.py
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""Adapter lookup.
|
|
2
|
+
|
|
3
|
+
Two ways to name an adapter:
|
|
4
|
+
|
|
5
|
+
--adapter demo-booking
|
|
6
|
+
one of the demo agents that ship with outturn
|
|
7
|
+
|
|
8
|
+
--adapter myproject.agents:MyAgent
|
|
9
|
+
anything importable. Your agent lives in your repo, not in this one.
|
|
10
|
+
|
|
11
|
+
The second form is the point. You should never have to edit a file inside
|
|
12
|
+
this package to test your own agent.
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import importlib
|
|
17
|
+
|
|
18
|
+
from .base import Agent
|
|
19
|
+
from .demo_booking import DemoBookingAgent
|
|
20
|
+
from .demo_order import DemoOrderAgent
|
|
21
|
+
|
|
22
|
+
BUILTIN: dict[str, type] = {
|
|
23
|
+
DemoOrderAgent.name: DemoOrderAgent,
|
|
24
|
+
DemoBookingAgent.name: DemoBookingAgent,
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class AdapterError(ValueError):
|
|
29
|
+
pass
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _load_path(spec: str) -> type:
|
|
33
|
+
"""Import 'package.module:ClassName' or 'package.module.ClassName'."""
|
|
34
|
+
if ":" in spec:
|
|
35
|
+
module_name, _, attr = spec.partition(":")
|
|
36
|
+
else:
|
|
37
|
+
module_name, _, attr = spec.rpartition(".")
|
|
38
|
+
if not module_name or not attr:
|
|
39
|
+
raise AdapterError(
|
|
40
|
+
f"{spec!r} is not a known adapter and does not look like an import path. "
|
|
41
|
+
f"Use 'module:ClassName', or one of: {', '.join(sorted(BUILTIN))}"
|
|
42
|
+
)
|
|
43
|
+
try:
|
|
44
|
+
module = importlib.import_module(module_name)
|
|
45
|
+
except ImportError as e:
|
|
46
|
+
raise AdapterError(f"could not import {module_name!r}: {e}") from e
|
|
47
|
+
try:
|
|
48
|
+
return getattr(module, attr)
|
|
49
|
+
except AttributeError as e:
|
|
50
|
+
raise AdapterError(f"{module_name!r} has no attribute {attr!r}") from e
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def get(spec: str) -> Agent:
|
|
54
|
+
"""Resolve an adapter name or import path into an instance."""
|
|
55
|
+
cls = BUILTIN.get(spec) or _load_path(spec)
|
|
56
|
+
|
|
57
|
+
agent = cls() if isinstance(cls, type) else cls # a factory function is fine too
|
|
58
|
+
|
|
59
|
+
for method in ("reset", "send", "outcome"):
|
|
60
|
+
if not callable(getattr(agent, method, None)):
|
|
61
|
+
raise AdapterError(
|
|
62
|
+
f"{spec!r} is missing {method}(). An adapter needs reset(), send(text) "
|
|
63
|
+
f"and outcome()."
|
|
64
|
+
)
|
|
65
|
+
return agent
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
__all__ = ["Agent", "AdapterError", "BUILTIN", "get", "DemoOrderAgent", "DemoBookingAgent"]
|
outturn/adapters/base.py
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
"""The adapter contract. Three methods.
|
|
2
|
+
|
|
3
|
+
reset() start a fresh conversation
|
|
4
|
+
send(text) one caller turn in, the agent's reply out
|
|
5
|
+
outcome() the structured result, right now
|
|
6
|
+
|
|
7
|
+
If your agent cannot hand back a structured outcome, outturn cannot help you,
|
|
8
|
+
and that is itself the finding. State that lives only inside the model's
|
|
9
|
+
context is not inspectable, and what is not inspectable is not testable.
|
|
10
|
+
Move the outcome into code and the tool works.
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from typing import Protocol, runtime_checkable
|
|
15
|
+
|
|
16
|
+
from ..models import Outcome
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@runtime_checkable
|
|
20
|
+
class Agent(Protocol):
|
|
21
|
+
name: str
|
|
22
|
+
|
|
23
|
+
def reset(self) -> None: ...
|
|
24
|
+
def send(self, text: str) -> str: ...
|
|
25
|
+
def outcome(self) -> Outcome: ...
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
"""A deliberately imperfect appointment booking agent.
|
|
2
|
+
|
|
3
|
+
This one exists to prove the point of the whole rewrite: a booking has no
|
|
4
|
+
line items at all. It is a handful of scalar fields. The same engine handles
|
|
5
|
+
it without a single special case.
|
|
6
|
+
|
|
7
|
+
Its planted flaw is the one real booking agents actually have: it hears the
|
|
8
|
+
service and the day, and forgets to carry the duration across when the
|
|
9
|
+
customer changes their mind mid call.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import random
|
|
14
|
+
import re
|
|
15
|
+
|
|
16
|
+
from ..models import Outcome
|
|
17
|
+
|
|
18
|
+
SERVICES = {
|
|
19
|
+
"deep tissue massage": 60,
|
|
20
|
+
"swedish massage": 60,
|
|
21
|
+
"sports massage": 90,
|
|
22
|
+
"facial": 45,
|
|
23
|
+
}
|
|
24
|
+
_DAYS = ("monday", "tuesday", "wednesday", "thursday", "friday", "saturday", "sunday")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class DemoBookingAgent:
|
|
28
|
+
name = "demo-booking"
|
|
29
|
+
|
|
30
|
+
def __init__(self, forget_duration_rate: float = 0.4) -> None:
|
|
31
|
+
self.forget_duration_rate = forget_duration_rate
|
|
32
|
+
self._state: dict = {}
|
|
33
|
+
|
|
34
|
+
def reset(self) -> None:
|
|
35
|
+
self._state = {"confirmed": False}
|
|
36
|
+
|
|
37
|
+
def send(self, text: str) -> str:
|
|
38
|
+
t = text.lower().strip()
|
|
39
|
+
changed = False
|
|
40
|
+
|
|
41
|
+
for service, minutes in SERVICES.items():
|
|
42
|
+
if service in t or service.split()[0] in t:
|
|
43
|
+
switching = "service" in self._state and self._state["service"] != service
|
|
44
|
+
self._state["service"] = service
|
|
45
|
+
# the planted flaw: on a switch it sometimes keeps the old duration
|
|
46
|
+
if not switching or random.random() > self.forget_duration_rate:
|
|
47
|
+
self._state["duration_minutes"] = minutes
|
|
48
|
+
changed = True
|
|
49
|
+
break
|
|
50
|
+
|
|
51
|
+
m = re.search(r"\b(\d{1,2})[:.](\d{2})\s*(am|pm)?\b", t)
|
|
52
|
+
if m:
|
|
53
|
+
hour, minute, ampm = int(m.group(1)), m.group(2), m.group(3)
|
|
54
|
+
if ampm == "pm" and hour < 12:
|
|
55
|
+
hour += 12
|
|
56
|
+
self._state["time"] = f"{hour:02d}:{minute}"
|
|
57
|
+
changed = True
|
|
58
|
+
|
|
59
|
+
for day in _DAYS:
|
|
60
|
+
if day in t:
|
|
61
|
+
self._state["day"] = day
|
|
62
|
+
changed = True
|
|
63
|
+
break
|
|
64
|
+
|
|
65
|
+
if any(w in t for w in ("yes", "confirm", "book it", "that works", "go ahead")):
|
|
66
|
+
self._state["confirmed"] = True
|
|
67
|
+
return "Booked. You will get a confirmation by text."
|
|
68
|
+
|
|
69
|
+
if changed:
|
|
70
|
+
bits = [str(self._state.get(k)) for k in ("service", "day", "time") if self._state.get(k)]
|
|
71
|
+
return "Got it, " + ", ".join(bits) + ". Shall I book that?"
|
|
72
|
+
return "Sorry, I did not catch that."
|
|
73
|
+
|
|
74
|
+
def outcome(self) -> Outcome:
|
|
75
|
+
return Outcome(entries=(), fields=dict(self._state))
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""A deliberately imperfect restaurant ordering agent.
|
|
2
|
+
|
|
3
|
+
It ships so you can see real output before writing any integration, and it
|
|
4
|
+
fails in two ways on purpose, because an eval you cannot fail is not an eval.
|
|
5
|
+
|
|
6
|
+
1. It drops a modifier at random under load. That is what FLAKY looks like.
|
|
7
|
+
2. Its scope guard is tuned too tight, so "do you sell garlic bread" is
|
|
8
|
+
treated as an off topic question rather than a customer trying to order.
|
|
9
|
+
That is not a strawman. It is behaviour observed on a shipped agent.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import random
|
|
14
|
+
import re
|
|
15
|
+
|
|
16
|
+
from ..models import Entry, Outcome
|
|
17
|
+
|
|
18
|
+
MENU = {
|
|
19
|
+
"pepperoni pizza": 12.00,
|
|
20
|
+
"margherita pizza": 10.50,
|
|
21
|
+
"garlic bread": 4.50,
|
|
22
|
+
"coke": 2.00,
|
|
23
|
+
"sparkling water": 2.50,
|
|
24
|
+
}
|
|
25
|
+
MODIFIERS = {"thin crust", "extra cheese", "no cheese", "well done", "sliced"}
|
|
26
|
+
_NUM = {"one": 1, "two": 2, "three": 3, "four": 4, "five": 5, "a": 1, "an": 1}
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class DemoOrderAgent:
|
|
30
|
+
name = "demo-order"
|
|
31
|
+
|
|
32
|
+
def __init__(self, drop_modifier_rate: float = 0.35) -> None:
|
|
33
|
+
self.drop_modifier_rate = drop_modifier_rate
|
|
34
|
+
self._lines: list[dict] = []
|
|
35
|
+
|
|
36
|
+
def reset(self) -> None:
|
|
37
|
+
self._lines = []
|
|
38
|
+
|
|
39
|
+
# ------------------------------------------------------------------ #
|
|
40
|
+
|
|
41
|
+
def _quantity(self, text: str) -> int:
|
|
42
|
+
m = re.search(r"\b(\d+)\b", text)
|
|
43
|
+
if m:
|
|
44
|
+
return int(m.group(1))
|
|
45
|
+
for word, n in _NUM.items():
|
|
46
|
+
if re.search(rf"\b{word}\b", text):
|
|
47
|
+
return n
|
|
48
|
+
return 1
|
|
49
|
+
|
|
50
|
+
def _find_item(self, text: str) -> str | None:
|
|
51
|
+
for item in sorted(MENU, key=len, reverse=True):
|
|
52
|
+
if item in text:
|
|
53
|
+
return item
|
|
54
|
+
if "pizza" in text:
|
|
55
|
+
return "pepperoni pizza"
|
|
56
|
+
return None
|
|
57
|
+
|
|
58
|
+
def send(self, text: str) -> str:
|
|
59
|
+
t = text.lower().strip()
|
|
60
|
+
|
|
61
|
+
# the scope guard, deliberately too tight
|
|
62
|
+
if t.startswith(("do you", "have you got", "is there")) and "?" in text:
|
|
63
|
+
return "I can only take orders from the menu, I am afraid."
|
|
64
|
+
|
|
65
|
+
if any(w in t for w in ("cancel", "remove", "take off", "drop the")):
|
|
66
|
+
target = self._find_item(t)
|
|
67
|
+
for i in range(len(self._lines) - 1, -1, -1):
|
|
68
|
+
if target is None or self._lines[i]["name"] == target:
|
|
69
|
+
removed = self._lines.pop(i)
|
|
70
|
+
return f"Removed the {removed['name']}."
|
|
71
|
+
return "There is nothing like that on your order."
|
|
72
|
+
|
|
73
|
+
mods = [m for m in MODIFIERS if m in t]
|
|
74
|
+
item = self._find_item(t)
|
|
75
|
+
|
|
76
|
+
# modifiers with no item named apply to the most recent line
|
|
77
|
+
if item is None and mods and self._lines:
|
|
78
|
+
line = self._lines[-1]
|
|
79
|
+
for m in mods:
|
|
80
|
+
if random.random() > self.drop_modifier_rate:
|
|
81
|
+
if m not in line["tags"]:
|
|
82
|
+
line["tags"].append(m)
|
|
83
|
+
return f"Noted on the {line['name']}."
|
|
84
|
+
|
|
85
|
+
if item is None:
|
|
86
|
+
if re.search(r"\bmake (it|that|those|them)\b", t):
|
|
87
|
+
if self._lines:
|
|
88
|
+
self._lines[-1]["quantity"] = self._quantity(t)
|
|
89
|
+
return f"Updated to {self._lines[-1]['quantity']}."
|
|
90
|
+
return "Sorry, I did not catch an item."
|
|
91
|
+
|
|
92
|
+
qty = self._quantity(t)
|
|
93
|
+
kept = [m for m in mods if random.random() > self.drop_modifier_rate]
|
|
94
|
+
self._lines.append({"name": item, "quantity": qty, "unit_price": MENU[item], "tags": kept})
|
|
95
|
+
return f"Added {qty} {item}."
|
|
96
|
+
|
|
97
|
+
def outcome(self) -> Outcome:
|
|
98
|
+
total = round(sum(l["quantity"] * l["unit_price"] for l in self._lines), 2)
|
|
99
|
+
return Outcome(
|
|
100
|
+
entries=tuple(
|
|
101
|
+
Entry(
|
|
102
|
+
name=l["name"],
|
|
103
|
+
numbers={"quantity": float(l["quantity"]), "unit_price": float(l["unit_price"])},
|
|
104
|
+
tags=tuple(l["tags"]),
|
|
105
|
+
)
|
|
106
|
+
for l in self._lines
|
|
107
|
+
),
|
|
108
|
+
fields={"total": total},
|
|
109
|
+
)
|
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
"""LiveKit Agents driver.
|
|
2
|
+
|
|
3
|
+
Not imported by default. Needs `pip install outturn[livekit]`.
|
|
4
|
+
|
|
5
|
+
LiveKit ships its own roomless testing path, and this driver uses it rather
|
|
6
|
+
than inventing one:
|
|
7
|
+
|
|
8
|
+
async with AgentSession(llm=llm) as session:
|
|
9
|
+
await session.start(MyAgent())
|
|
10
|
+
result = await session.run(user_input="two pepperoni pizzas")
|
|
11
|
+
|
|
12
|
+
No room, no audio, no telephony. Fast enough to run in CI on every commit.
|
|
13
|
+
|
|
14
|
+
THE ONE DESIGN DECISION WORTH KNOWING
|
|
15
|
+
|
|
16
|
+
The outcome does not live in LiveKit. It lives in whatever your own function
|
|
17
|
+
tools wrote it to, usually the session userdata, and only you know its shape.
|
|
18
|
+
So this driver does the session work and you pass one callable that reads
|
|
19
|
+
your state and returns an Outcome. The driver stays thin and never breaks
|
|
20
|
+
because someone structured their state differently.
|
|
21
|
+
|
|
22
|
+
USAGE
|
|
23
|
+
|
|
24
|
+
from dataclasses import dataclass, field
|
|
25
|
+
from livekit.agents import AgentSession, inference
|
|
26
|
+
from outturn.adapters.livekit import LiveKitAgent
|
|
27
|
+
from outturn.models import Outcome
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class Booking:
|
|
31
|
+
service: str | None = None
|
|
32
|
+
day: str | None = None
|
|
33
|
+
confirmed: bool = False
|
|
34
|
+
|
|
35
|
+
async def make_session():
|
|
36
|
+
llm = inference.LLM(model="openai/gpt-4o-mini")
|
|
37
|
+
state = Booking()
|
|
38
|
+
session = AgentSession(llm=llm, userdata=state)
|
|
39
|
+
return session, MyAgent(), state
|
|
40
|
+
|
|
41
|
+
def read_outcome(state: Booking) -> Outcome:
|
|
42
|
+
return Outcome(fields={
|
|
43
|
+
"service": state.service,
|
|
44
|
+
"day": state.day,
|
|
45
|
+
"confirmed": state.confirmed,
|
|
46
|
+
})
|
|
47
|
+
|
|
48
|
+
agent = LiveKitAgent(make_session, read_outcome)
|
|
49
|
+
|
|
50
|
+
# then: outturn scenarios/ --adapter myproject.evals:agent
|
|
51
|
+
|
|
52
|
+
WHAT THIS DOES NOT TEST
|
|
53
|
+
|
|
54
|
+
Speech to text, endpointing, turn taking. It measures whether your agent
|
|
55
|
+
understands correctly, not whether it hears correctly. Both matter.
|
|
56
|
+
"""
|
|
57
|
+
from __future__ import annotations
|
|
58
|
+
|
|
59
|
+
import asyncio
|
|
60
|
+
import inspect
|
|
61
|
+
from typing import Any, Callable
|
|
62
|
+
|
|
63
|
+
from ..models import Outcome
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class LiveKitAgent:
|
|
69
|
+
"""Drives a LiveKit AgentSession through text turns, one scenario at a time.
|
|
70
|
+
|
|
71
|
+
session_factory async callable returning (session, agent, state).
|
|
72
|
+
Called fresh on every reset, so each run starts clean.
|
|
73
|
+
read_outcome takes that state, returns an Outcome.
|
|
74
|
+
timeout seconds to wait for a single turn before the run fails.
|
|
75
|
+
"""
|
|
76
|
+
|
|
77
|
+
name = "livekit"
|
|
78
|
+
|
|
79
|
+
def __init__(
|
|
80
|
+
self,
|
|
81
|
+
session_factory: Callable[[], Any],
|
|
82
|
+
read_outcome: Callable[[Any], Outcome],
|
|
83
|
+
*,
|
|
84
|
+
timeout: float = 60.0,
|
|
85
|
+
) -> None:
|
|
86
|
+
self._factory = session_factory
|
|
87
|
+
self._read = read_outcome
|
|
88
|
+
self._timeout = timeout
|
|
89
|
+
self._session: Any = None
|
|
90
|
+
self._state: Any = None
|
|
91
|
+
self._loop: asyncio.AbstractEventLoop | None = None
|
|
92
|
+
|
|
93
|
+
# -- event loop plumbing, because outturn's contract is sync ----------- #
|
|
94
|
+
|
|
95
|
+
def _get_loop(self) -> asyncio.AbstractEventLoop:
|
|
96
|
+
if self._loop is None or self._loop.is_closed():
|
|
97
|
+
self._loop = asyncio.new_event_loop()
|
|
98
|
+
asyncio.set_event_loop(self._loop)
|
|
99
|
+
return self._loop
|
|
100
|
+
|
|
101
|
+
def _run(self, coro):
|
|
102
|
+
return self._get_loop().run_until_complete(coro)
|
|
103
|
+
|
|
104
|
+
# -- the three method contract ---------------------------------------- #
|
|
105
|
+
|
|
106
|
+
def reset(self) -> None:
|
|
107
|
+
"""Fresh session for every run. No state leaks between runs."""
|
|
108
|
+
self.close()
|
|
109
|
+
|
|
110
|
+
async def _start():
|
|
111
|
+
made = self._factory()
|
|
112
|
+
if inspect.isawaitable(made):
|
|
113
|
+
made = await made
|
|
114
|
+
if not isinstance(made, tuple) or len(made) != 3:
|
|
115
|
+
raise TypeError(
|
|
116
|
+
"session_factory must return (session, agent, state), "
|
|
117
|
+
f"got {type(made).__name__}"
|
|
118
|
+
)
|
|
119
|
+
session, agent, state = made
|
|
120
|
+
await session.__aenter__()
|
|
121
|
+
await session.start(agent)
|
|
122
|
+
return session, state
|
|
123
|
+
|
|
124
|
+
self._session, self._state = self._run(_start())
|
|
125
|
+
|
|
126
|
+
def send(self, text: str) -> str:
|
|
127
|
+
if self._session is None:
|
|
128
|
+
raise RuntimeError("call reset() before send()")
|
|
129
|
+
|
|
130
|
+
async def _turn() -> str:
|
|
131
|
+
result = await asyncio.wait_for(
|
|
132
|
+
self._session.run(user_input=text), timeout=self._timeout
|
|
133
|
+
)
|
|
134
|
+
return _extract_reply(result)
|
|
135
|
+
|
|
136
|
+
return self._run(_turn())
|
|
137
|
+
|
|
138
|
+
def outcome(self) -> Outcome:
|
|
139
|
+
if self._state is None:
|
|
140
|
+
raise RuntimeError("call reset() before outcome()")
|
|
141
|
+
return self._read(self._state)
|
|
142
|
+
|
|
143
|
+
def close(self) -> None:
|
|
144
|
+
if self._session is not None and self._loop and not self._loop.is_closed():
|
|
145
|
+
try:
|
|
146
|
+
self._run(self._session.__aexit__(None, None, None))
|
|
147
|
+
except Exception:
|
|
148
|
+
pass
|
|
149
|
+
self._session = None
|
|
150
|
+
self._state = None
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _extract_reply(result: Any) -> str:
|
|
154
|
+
"""Extract the assistant's text from a RunResult.
|
|
155
|
+
|
|
156
|
+
Verified against livekit-agents 1.8.2: the reply is a ChatMessageEvent
|
|
157
|
+
in result.events with item.role == 'assistant' and item.content as a list
|
|
158
|
+
of strings. Returns the last such message, or "" if none is found.
|
|
159
|
+
|
|
160
|
+
Returns "" rather than raising so the outcome assertion can still run.
|
|
161
|
+
expect_reply_contains is meant to be used sparingly anyway.
|
|
162
|
+
"""
|
|
163
|
+
events = getattr(result, "events", None) or []
|
|
164
|
+
for event in reversed(list(events)):
|
|
165
|
+
item = getattr(event, "item", None)
|
|
166
|
+
if item is None:
|
|
167
|
+
continue
|
|
168
|
+
if getattr(item, "role", None) != "assistant":
|
|
169
|
+
continue
|
|
170
|
+
content = getattr(item, "content", None)
|
|
171
|
+
if isinstance(content, list) and content:
|
|
172
|
+
first = content[0]
|
|
173
|
+
if isinstance(first, str) and first:
|
|
174
|
+
return first
|
|
175
|
+
return ""
|
outturn/assertions.py
ADDED
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
"""The diff engine.
|
|
2
|
+
|
|
3
|
+
One rule governs every line here: numbers are compared exactly, words are
|
|
4
|
+
compared fuzzily. A quantity of 3 is not "close to" 2 and a duration of 90
|
|
5
|
+
minutes is not "close to" 60. But "deep tissue massage" and "Deep Tissue
|
|
6
|
+
Massage (60min)" are the same service.
|
|
7
|
+
|
|
8
|
+
Everywhere being clever would have been possible, being clever would have
|
|
9
|
+
hidden a real failure behind a generous match. That is the one thing a test
|
|
10
|
+
harness must never do.
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import re
|
|
15
|
+
from difflib import SequenceMatcher
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
from .models import Discrepancy, Entry, Kind, Outcome
|
|
19
|
+
|
|
20
|
+
_PUNCT = re.compile(r"[^a-z0-9 ]+")
|
|
21
|
+
_SPACE = re.compile(r"\s+")
|
|
22
|
+
# words carrying no identity, dropped before comparison
|
|
23
|
+
_NOISE = {"a", "an", "the", "of", "with", "and", "please", "order", "for"}
|
|
24
|
+
|
|
25
|
+
DEFAULT_NAME_THRESHOLD = 0.82
|
|
26
|
+
DEFAULT_TAG_THRESHOLD = 0.80
|
|
27
|
+
DEFAULT_TOLERANCE = 0.005
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def normalize(text: str) -> str:
|
|
31
|
+
"""Lowercase, strip punctuation, drop noise words, crudely singularize.
|
|
32
|
+
|
|
33
|
+
Deliberately simple. If you need a stemmer here, your names are the
|
|
34
|
+
problem, not the matcher.
|
|
35
|
+
"""
|
|
36
|
+
t = _PUNCT.sub(" ", str(text).lower())
|
|
37
|
+
words = [w for w in _SPACE.sub(" ", t).strip().split(" ") if w and w not in _NOISE]
|
|
38
|
+
out = []
|
|
39
|
+
for w in words:
|
|
40
|
+
if len(w) > 3 and w.endswith("es") and not w.endswith("ses"):
|
|
41
|
+
w = w[:-2]
|
|
42
|
+
elif len(w) > 3 and w.endswith("s") and not w.endswith("ss"):
|
|
43
|
+
w = w[:-1]
|
|
44
|
+
out.append(w)
|
|
45
|
+
return " ".join(sorted(out))
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def similarity(a: str, b: str) -> float:
|
|
49
|
+
return SequenceMatcher(None, normalize(a), normalize(b)).ratio()
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _numbers_equal(a: float, b: float, tolerance: float) -> bool:
|
|
53
|
+
return abs(float(a) - float(b)) <= tolerance
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _number_agreement(want: Entry, have: Entry, tolerance: float) -> int:
|
|
57
|
+
"""How many of the expected numbers this candidate also matches.
|
|
58
|
+
|
|
59
|
+
Used only to break ties between candidates whose names score identically.
|
|
60
|
+
Two cart lines called "coffee" are indistinguishable by name, so the one
|
|
61
|
+
whose quantity also matches is the intended pairing. vaevals picked
|
|
62
|
+
whichever came first, which made a real bug look like two.
|
|
63
|
+
"""
|
|
64
|
+
return sum(
|
|
65
|
+
1
|
|
66
|
+
for k, v in want.numbers.items()
|
|
67
|
+
if k in have.numbers and _numbers_equal(v, have.numbers[k], tolerance)
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _match_tags(
|
|
72
|
+
expected: tuple[str, ...],
|
|
73
|
+
actual: tuple[str, ...],
|
|
74
|
+
where: str,
|
|
75
|
+
threshold: float,
|
|
76
|
+
) -> list[Discrepancy]:
|
|
77
|
+
out: list[Discrepancy] = []
|
|
78
|
+
remaining = list(actual)
|
|
79
|
+
for want in expected:
|
|
80
|
+
best_i, best_score = -1, 0.0
|
|
81
|
+
for i, have in enumerate(remaining):
|
|
82
|
+
s = similarity(want, have)
|
|
83
|
+
if s > best_score:
|
|
84
|
+
best_i, best_score = i, s
|
|
85
|
+
if best_score >= threshold:
|
|
86
|
+
remaining.pop(best_i)
|
|
87
|
+
else:
|
|
88
|
+
out.append(Discrepancy(Kind.MISSING_TAG, where, expected=want, actual=None))
|
|
89
|
+
for leftover in remaining:
|
|
90
|
+
out.append(Discrepancy(Kind.EXTRA_TAG, where, expected=None, actual=leftover))
|
|
91
|
+
return out
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _compare_fields(
|
|
95
|
+
expected: dict[str, Any],
|
|
96
|
+
actual: dict[str, Any],
|
|
97
|
+
*,
|
|
98
|
+
tag_threshold: float,
|
|
99
|
+
tolerance: float,
|
|
100
|
+
) -> list[Discrepancy]:
|
|
101
|
+
"""Scalars. Numbers and booleans exact, strings fuzzy.
|
|
102
|
+
|
|
103
|
+
Extra fields in the actual outcome are ignored on purpose. The scenario
|
|
104
|
+
states what must be true, not everything that may be present.
|
|
105
|
+
"""
|
|
106
|
+
out: list[Discrepancy] = []
|
|
107
|
+
for key, want in expected.items():
|
|
108
|
+
if key not in actual or actual[key] is None:
|
|
109
|
+
out.append(Discrepancy(Kind.MISSING_FIELD, key, expected=want, actual=None))
|
|
110
|
+
continue
|
|
111
|
+
have = actual[key]
|
|
112
|
+
|
|
113
|
+
if isinstance(want, bool) or isinstance(have, bool):
|
|
114
|
+
if bool(want) != bool(have):
|
|
115
|
+
out.append(Discrepancy(Kind.WRONG_FIELD, key, expected=want, actual=have))
|
|
116
|
+
elif isinstance(want, (int, float)) and isinstance(have, (int, float)):
|
|
117
|
+
if not _numbers_equal(want, have, tolerance):
|
|
118
|
+
out.append(Discrepancy(Kind.WRONG_FIELD, key, expected=want, actual=have))
|
|
119
|
+
else:
|
|
120
|
+
if similarity(str(want), str(have)) < tag_threshold:
|
|
121
|
+
out.append(Discrepancy(Kind.WRONG_FIELD, key, expected=want, actual=have))
|
|
122
|
+
return out
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def compare_outcomes(
|
|
126
|
+
expected: Outcome,
|
|
127
|
+
actual: Outcome,
|
|
128
|
+
*,
|
|
129
|
+
name_threshold: float = DEFAULT_NAME_THRESHOLD,
|
|
130
|
+
tag_threshold: float = DEFAULT_TAG_THRESHOLD,
|
|
131
|
+
tolerance: float = DEFAULT_TOLERANCE,
|
|
132
|
+
) -> list[Discrepancy]:
|
|
133
|
+
"""Best-match on entry names, exact on every number, fuzzy on every word."""
|
|
134
|
+
out: list[Discrepancy] = []
|
|
135
|
+
unmatched: list[Entry] = list(actual.entries)
|
|
136
|
+
|
|
137
|
+
for want in expected.entries:
|
|
138
|
+
best_i, best_key = -1, (0.0, -1)
|
|
139
|
+
for i, have in enumerate(unmatched):
|
|
140
|
+
score = similarity(want.name, have.name)
|
|
141
|
+
key = (score, _number_agreement(want, have, tolerance))
|
|
142
|
+
if key > best_key:
|
|
143
|
+
best_i, best_key = i, key
|
|
144
|
+
|
|
145
|
+
if best_key[0] < name_threshold:
|
|
146
|
+
out.append(
|
|
147
|
+
Discrepancy(Kind.MISSING_ENTRY, "not found", expected=want.name, actual=None)
|
|
148
|
+
)
|
|
149
|
+
continue
|
|
150
|
+
|
|
151
|
+
have = unmatched.pop(best_i)
|
|
152
|
+
|
|
153
|
+
for key, wanted in want.numbers.items():
|
|
154
|
+
if key not in have.numbers:
|
|
155
|
+
out.append(
|
|
156
|
+
Discrepancy(Kind.MISSING_NUMBER, f"{key} on {want.name!r}",
|
|
157
|
+
expected=wanted, actual=None)
|
|
158
|
+
)
|
|
159
|
+
elif not _numbers_equal(wanted, have.numbers[key], tolerance):
|
|
160
|
+
out.append(
|
|
161
|
+
Discrepancy(Kind.WRONG_NUMBER, f"{key} on {want.name!r}",
|
|
162
|
+
expected=wanted, actual=have.numbers[key])
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
out.extend(_match_tags(want.tags, have.tags, f"on {want.name!r}", tag_threshold))
|
|
166
|
+
|
|
167
|
+
for leftover in unmatched:
|
|
168
|
+
out.append(
|
|
169
|
+
Discrepancy(Kind.EXTRA_ENTRY, "not expected", expected=None, actual=leftover.name)
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
out.extend(
|
|
173
|
+
_compare_fields(expected.fields, actual.fields,
|
|
174
|
+
tag_threshold=tag_threshold, tolerance=tolerance)
|
|
175
|
+
)
|
|
176
|
+
return out
|
outturn/cli.py
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Command line entry point."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import random
|
|
6
|
+
import sys
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from . import adapters, report, runner, scenario
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
13
|
+
p = argparse.ArgumentParser(
|
|
14
|
+
prog="outturn",
|
|
15
|
+
description="Did the agent produce the right outcome? As a number.",
|
|
16
|
+
)
|
|
17
|
+
p.add_argument("path", type=Path, help="a scenario file, or a directory of them")
|
|
18
|
+
p.add_argument("--adapter", default="demo-order",
|
|
19
|
+
help="a built in name, or an import path like myproject.agents:MyAgent")
|
|
20
|
+
p.add_argument("--runs", type=int, default=5, help="runs per scenario (default 5)")
|
|
21
|
+
p.add_argument("--seed", type=int, default=None, help="seed the RNG for a repeatable run")
|
|
22
|
+
p.add_argument("--threshold", type=float, default=None,
|
|
23
|
+
help="exit 1 if overall accuracy falls below this (0 to 1)")
|
|
24
|
+
p.add_argument("--json", action="store_true", help="machine readable output")
|
|
25
|
+
p.add_argument("--list-adapters", action="store_true", help="show registered adapters and exit")
|
|
26
|
+
return p
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def main(argv: list[str] | None = None) -> int:
|
|
30
|
+
args = build_parser().parse_args(argv)
|
|
31
|
+
|
|
32
|
+
if args.list_adapters:
|
|
33
|
+
for name in sorted(adapters.BUILTIN):
|
|
34
|
+
print(name)
|
|
35
|
+
print("\nor any import path, for example myproject.agents:MyAgent")
|
|
36
|
+
return 0
|
|
37
|
+
|
|
38
|
+
if args.runs < 1:
|
|
39
|
+
print("--runs must be at least 1", file=sys.stderr)
|
|
40
|
+
return 2
|
|
41
|
+
if args.seed is not None:
|
|
42
|
+
random.seed(args.seed)
|
|
43
|
+
|
|
44
|
+
try:
|
|
45
|
+
scenarios = scenario.load(args.path)
|
|
46
|
+
agent = adapters.get(args.adapter)
|
|
47
|
+
except (scenario.ScenarioError, adapters.AdapterError, OSError) as e:
|
|
48
|
+
print(str(e).strip("'"), file=sys.stderr)
|
|
49
|
+
return 2
|
|
50
|
+
|
|
51
|
+
reports = runner.run_all(agent, scenarios, args.runs)
|
|
52
|
+
print(report.as_json(reports) if args.json else report.render(reports))
|
|
53
|
+
|
|
54
|
+
if args.threshold is not None:
|
|
55
|
+
passes, runs = report.overall(reports)
|
|
56
|
+
if runs and (passes / runs) < args.threshold:
|
|
57
|
+
return 1
|
|
58
|
+
return 0
|
outturn/models.py
ADDED
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
"""Core data types.
|
|
2
|
+
|
|
3
|
+
One idea runs through all of it: an agent conversation ends in a structured
|
|
4
|
+
outcome, and a structured outcome can be compared exactly. A restaurant order
|
|
5
|
+
is one kind of outcome. So is a booking, a triage, a qualified lead.
|
|
6
|
+
|
|
7
|
+
Two comparison rules, applied everywhere:
|
|
8
|
+
numbers are compared exactly a quantity of 3 is not close to 2
|
|
9
|
+
words are compared fuzzily "deep tissue" and "Deep Tissue Massage" match
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
from enum import Enum
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass(frozen=True)
|
|
19
|
+
class Entry:
|
|
20
|
+
"""One repeated thing in an outcome.
|
|
21
|
+
|
|
22
|
+
A cart line. A passenger. A prescription. Anything the agent can produce
|
|
23
|
+
more than one of.
|
|
24
|
+
|
|
25
|
+
name matched fuzzily, it is how the entry is identified
|
|
26
|
+
numbers matched exactly, quantity, price, duration, dosage
|
|
27
|
+
tags matched fuzzily as a set, modifiers, options, flags
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
name: str
|
|
31
|
+
numbers: dict[str, float] = field(default_factory=dict)
|
|
32
|
+
tags: tuple[str, ...] = ()
|
|
33
|
+
|
|
34
|
+
@staticmethod
|
|
35
|
+
def from_dict(d: dict[str, Any]) -> "Entry":
|
|
36
|
+
if "name" not in d:
|
|
37
|
+
raise ValueError("an entry needs a 'name'")
|
|
38
|
+
numbers = {
|
|
39
|
+
str(k): float(v)
|
|
40
|
+
for k, v in d.items()
|
|
41
|
+
if k not in ("name", "tags") and isinstance(v, (int, float)) and not isinstance(v, bool)
|
|
42
|
+
}
|
|
43
|
+
tags = d.get("tags", ())
|
|
44
|
+
if isinstance(tags, str):
|
|
45
|
+
tags = (tags,)
|
|
46
|
+
return Entry(name=str(d["name"]), numbers=numbers, tags=tuple(str(t) for t in tags))
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass(frozen=True)
|
|
50
|
+
class Outcome:
|
|
51
|
+
"""What the conversation actually produced.
|
|
52
|
+
|
|
53
|
+
entries the repeated things, may be empty
|
|
54
|
+
fields the scalars. numbers and booleans exact, strings fuzzy
|
|
55
|
+
|
|
56
|
+
A restaurant order is entries plus a total field.
|
|
57
|
+
A booking is no entries at all, just fields.
|
|
58
|
+
Both work.
|
|
59
|
+
"""
|
|
60
|
+
|
|
61
|
+
entries: tuple[Entry, ...] = ()
|
|
62
|
+
fields: dict[str, Any] = field(default_factory=dict)
|
|
63
|
+
|
|
64
|
+
@staticmethod
|
|
65
|
+
def from_dict(d: dict[str, Any] | None) -> "Outcome":
|
|
66
|
+
d = dict(d or {})
|
|
67
|
+
raw_entries = d.pop("entries", ())
|
|
68
|
+
return Outcome(
|
|
69
|
+
entries=tuple(Entry.from_dict(e) for e in raw_entries),
|
|
70
|
+
fields={str(k): v for k, v in d.items()},
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
def is_empty(self) -> bool:
|
|
74
|
+
return not self.entries and not self.fields
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@dataclass(frozen=True)
|
|
78
|
+
class Turn:
|
|
79
|
+
"""One thing the caller says."""
|
|
80
|
+
|
|
81
|
+
user: str
|
|
82
|
+
# substrings the reply must contain. use sparingly: asserting on phrasing
|
|
83
|
+
# is brittle by nature, which is the problem this tool exists to avoid.
|
|
84
|
+
expect_reply_contains: tuple[str, ...] = ()
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
@dataclass(frozen=True)
|
|
88
|
+
class Scenario:
|
|
89
|
+
id: str
|
|
90
|
+
turns: tuple[Turn, ...]
|
|
91
|
+
expect: Outcome
|
|
92
|
+
description: str = ""
|
|
93
|
+
tags: tuple[str, ...] = ()
|
|
94
|
+
runs: int | None = None # overrides the global run count
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
class Kind(str, Enum):
|
|
98
|
+
MISSING_ENTRY = "missing_entry"
|
|
99
|
+
EXTRA_ENTRY = "extra_entry"
|
|
100
|
+
WRONG_NUMBER = "wrong_number"
|
|
101
|
+
MISSING_NUMBER = "missing_number"
|
|
102
|
+
MISSING_TAG = "missing_tag"
|
|
103
|
+
EXTRA_TAG = "extra_tag"
|
|
104
|
+
MISSING_FIELD = "missing_field"
|
|
105
|
+
WRONG_FIELD = "wrong_field"
|
|
106
|
+
REPLY_MISSING_TEXT = "reply_missing_text"
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
@dataclass(frozen=True)
|
|
110
|
+
class Discrepancy:
|
|
111
|
+
kind: Kind
|
|
112
|
+
detail: str
|
|
113
|
+
expected: Any = None
|
|
114
|
+
actual: Any = None
|
|
115
|
+
|
|
116
|
+
def __str__(self) -> str:
|
|
117
|
+
if self.expected is None and self.actual is None:
|
|
118
|
+
return f"{self.kind.value}: {self.detail}"
|
|
119
|
+
return f"{self.kind.value}: {self.detail} (expected {self.expected!r}, got {self.actual!r})"
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
@dataclass
|
|
123
|
+
class RunResult:
|
|
124
|
+
scenario_id: str
|
|
125
|
+
run_index: int
|
|
126
|
+
discrepancies: list[Discrepancy] = field(default_factory=list)
|
|
127
|
+
transcript: list[tuple[str, str]] = field(default_factory=list)
|
|
128
|
+
error: str | None = None
|
|
129
|
+
|
|
130
|
+
@property
|
|
131
|
+
def passed(self) -> bool:
|
|
132
|
+
return self.error is None and not self.discrepancies
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
@dataclass
|
|
136
|
+
class ScenarioReport:
|
|
137
|
+
scenario_id: str
|
|
138
|
+
results: list[RunResult] = field(default_factory=list)
|
|
139
|
+
|
|
140
|
+
@property
|
|
141
|
+
def runs(self) -> int:
|
|
142
|
+
return len(self.results)
|
|
143
|
+
|
|
144
|
+
@property
|
|
145
|
+
def passes(self) -> int:
|
|
146
|
+
return sum(1 for r in self.results if r.passed)
|
|
147
|
+
|
|
148
|
+
@property
|
|
149
|
+
def pass_rate(self) -> float:
|
|
150
|
+
return (self.passes / self.runs) if self.runs else 0.0
|
|
151
|
+
|
|
152
|
+
@property
|
|
153
|
+
def is_flaky(self) -> bool:
|
|
154
|
+
"""Passed sometimes and failed sometimes.
|
|
155
|
+
|
|
156
|
+
Worse than always failing, because it ships.
|
|
157
|
+
"""
|
|
158
|
+
return 0 < self.passes < self.runs
|
outturn/report.py
ADDED
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"""Turns reports into something a person reads in five seconds."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
from collections import Counter
|
|
6
|
+
|
|
7
|
+
from .models import ScenarioReport
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _failure_lines(rep: ScenarioReport, limit: int = 3) -> list[str]:
|
|
11
|
+
counts: Counter[str] = Counter()
|
|
12
|
+
for r in rep.results:
|
|
13
|
+
if r.error:
|
|
14
|
+
counts[f"error: {r.error}"] += 1
|
|
15
|
+
for d in r.discrepancies:
|
|
16
|
+
counts[str(d)] += 1
|
|
17
|
+
return [f" {n}x {text}" for text, n in counts.most_common(limit)]
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def overall(reports: list[ScenarioReport]) -> tuple[int, int]:
|
|
21
|
+
passes = sum(r.passes for r in reports)
|
|
22
|
+
runs = sum(r.runs for r in reports)
|
|
23
|
+
return passes, runs
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def render(reports: list[ScenarioReport]) -> str:
|
|
27
|
+
lines: list[str] = []
|
|
28
|
+
for rep in reports:
|
|
29
|
+
flag = " FLAKY" if rep.is_flaky else ""
|
|
30
|
+
lines.append(
|
|
31
|
+
f"{rep.pass_rate * 100:5.0f}% {rep.scenario_id:<34} {rep.passes}/{rep.runs}{flag}"
|
|
32
|
+
)
|
|
33
|
+
if rep.passes < rep.runs:
|
|
34
|
+
lines.extend(_failure_lines(rep))
|
|
35
|
+
|
|
36
|
+
passes, runs = overall(reports)
|
|
37
|
+
rate = (passes / runs * 100) if runs else 0.0
|
|
38
|
+
lines.append("")
|
|
39
|
+
lines.append(f"outcome accuracy: {rate:.1f}% ({passes}/{runs} runs)")
|
|
40
|
+
|
|
41
|
+
flaky = [r.scenario_id for r in reports if r.is_flaky]
|
|
42
|
+
if flaky:
|
|
43
|
+
lines.append(f"flaky scenarios ({len(flaky)}): {', '.join(flaky)}")
|
|
44
|
+
lines.append("a scenario that passes sometimes is a bug that ships sometimes.")
|
|
45
|
+
return "\n".join(lines)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def as_json(reports: list[ScenarioReport]) -> str:
|
|
49
|
+
passes, runs = overall(reports)
|
|
50
|
+
return json.dumps(
|
|
51
|
+
{
|
|
52
|
+
"accuracy": (passes / runs) if runs else 0.0,
|
|
53
|
+
"passes": passes,
|
|
54
|
+
"runs": runs,
|
|
55
|
+
"flaky": [r.scenario_id for r in reports if r.is_flaky],
|
|
56
|
+
"scenarios": [
|
|
57
|
+
{
|
|
58
|
+
"id": r.scenario_id,
|
|
59
|
+
"passes": r.passes,
|
|
60
|
+
"runs": r.runs,
|
|
61
|
+
"pass_rate": r.pass_rate,
|
|
62
|
+
"flaky": r.is_flaky,
|
|
63
|
+
"failures": sorted(
|
|
64
|
+
{str(d) for res in r.results for d in res.discrepancies}
|
|
65
|
+
),
|
|
66
|
+
"errors": sorted({res.error for res in r.results if res.error}),
|
|
67
|
+
}
|
|
68
|
+
for r in reports
|
|
69
|
+
],
|
|
70
|
+
},
|
|
71
|
+
indent=2,
|
|
72
|
+
)
|
outturn/runner.py
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""Runs each scenario N times.
|
|
2
|
+
|
|
3
|
+
N, not once. Agents are nondeterministic, so one green run is not evidence.
|
|
4
|
+
A bug that appears one run in five is invisible to a boolean test and obvious
|
|
5
|
+
in a percentage.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from .adapters.base import Agent
|
|
10
|
+
from .assertions import compare_outcomes
|
|
11
|
+
from .models import Discrepancy, Kind, RunResult, Scenario, ScenarioReport
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def run_once(agent: Agent, scenario: Scenario, run_index: int) -> RunResult:
|
|
15
|
+
result = RunResult(scenario_id=scenario.id, run_index=run_index)
|
|
16
|
+
try:
|
|
17
|
+
agent.reset()
|
|
18
|
+
for turn in scenario.turns:
|
|
19
|
+
reply = agent.send(turn.user)
|
|
20
|
+
result.transcript.append((turn.user, reply))
|
|
21
|
+
for needle in turn.expect_reply_contains:
|
|
22
|
+
if needle.lower() not in (reply or "").lower():
|
|
23
|
+
result.discrepancies.append(
|
|
24
|
+
Discrepancy(Kind.REPLY_MISSING_TEXT, f"after {turn.user!r}",
|
|
25
|
+
expected=needle, actual=reply)
|
|
26
|
+
)
|
|
27
|
+
result.discrepancies.extend(compare_outcomes(scenario.expect, agent.outcome()))
|
|
28
|
+
except Exception as e: # an agent that crashes is a failing run, not a crashing suite
|
|
29
|
+
result.error = f"{type(e).__name__}: {e}"
|
|
30
|
+
return result
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def run_scenario(agent: Agent, scenario: Scenario, runs: int) -> ScenarioReport:
|
|
34
|
+
n = scenario.runs if scenario.runs is not None else runs
|
|
35
|
+
report = ScenarioReport(scenario_id=scenario.id)
|
|
36
|
+
for i in range(n):
|
|
37
|
+
report.results.append(run_once(agent, scenario, i))
|
|
38
|
+
return report
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def run_all(agent: Agent, scenarios: list[Scenario], runs: int) -> list[ScenarioReport]:
|
|
42
|
+
return [run_scenario(agent, s, runs) for s in scenarios]
|
outturn/scenario.py
ADDED
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""Load scenarios from YAML. Fail loudly on a bad file, never guess."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
import yaml
|
|
7
|
+
|
|
8
|
+
from .models import Outcome, Scenario, Turn
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class ScenarioError(ValueError):
|
|
12
|
+
pass
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _turn(raw: object, ctx: str) -> Turn:
|
|
16
|
+
if isinstance(raw, str):
|
|
17
|
+
return Turn(user=raw)
|
|
18
|
+
if isinstance(raw, dict):
|
|
19
|
+
if "user" not in raw:
|
|
20
|
+
raise ScenarioError(f"{ctx}: a turn needs a 'user' key")
|
|
21
|
+
contains = raw.get("expect_reply_contains", ())
|
|
22
|
+
if isinstance(contains, str):
|
|
23
|
+
contains = (contains,)
|
|
24
|
+
return Turn(user=str(raw["user"]), expect_reply_contains=tuple(str(c) for c in contains))
|
|
25
|
+
raise ScenarioError(f"{ctx}: a turn must be a string or a mapping, got {type(raw).__name__}")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def load_file(path: Path) -> Scenario:
|
|
29
|
+
raw = yaml.safe_load(path.read_text())
|
|
30
|
+
if not isinstance(raw, dict):
|
|
31
|
+
raise ScenarioError(f"{path}: top level must be a mapping")
|
|
32
|
+
for key in ("id", "turns", "expect"):
|
|
33
|
+
if key not in raw:
|
|
34
|
+
raise ScenarioError(f"{path}: missing required key {key!r}")
|
|
35
|
+
turns = raw["turns"]
|
|
36
|
+
if not isinstance(turns, list) or not turns:
|
|
37
|
+
raise ScenarioError(f"{path}: 'turns' must be a non empty list")
|
|
38
|
+
try:
|
|
39
|
+
expect = Outcome.from_dict(raw["expect"])
|
|
40
|
+
except ValueError as e:
|
|
41
|
+
raise ScenarioError(f"{path}: bad 'expect' block, {e}") from e
|
|
42
|
+
if expect.is_empty():
|
|
43
|
+
raise ScenarioError(f"{path}: 'expect' asserts nothing. A scenario that cannot fail is not a test.")
|
|
44
|
+
return Scenario(
|
|
45
|
+
id=str(raw["id"]),
|
|
46
|
+
description=str(raw.get("description", "")),
|
|
47
|
+
turns=tuple(_turn(t, str(path)) for t in turns),
|
|
48
|
+
expect=expect,
|
|
49
|
+
tags=tuple(str(t) for t in raw.get("tags", ())),
|
|
50
|
+
runs=(int(raw["runs"]) if raw.get("runs") is not None else None),
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def load(target: Path) -> list[Scenario]:
|
|
55
|
+
"""A single file, or every .yaml under a directory, recursively."""
|
|
56
|
+
if target.is_file():
|
|
57
|
+
return [load_file(target)]
|
|
58
|
+
files = sorted(p for p in target.rglob("*.y*ml") if p.is_file())
|
|
59
|
+
if not files:
|
|
60
|
+
raise ScenarioError(f"{target}: no .yaml scenario files found")
|
|
61
|
+
scenarios = [load_file(p) for p in files]
|
|
62
|
+
seen: set[str] = set()
|
|
63
|
+
for s in scenarios:
|
|
64
|
+
if s.id in seen:
|
|
65
|
+
raise ScenarioError(f"duplicate scenario id {s.id!r}")
|
|
66
|
+
seen.add(s.id)
|
|
67
|
+
return scenarios
|
|
@@ -0,0 +1,240 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: outturn
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Did the agent produce the right outcome? As a number.
|
|
5
|
+
Author: Wisdom Omons
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/OsasDTEch/outturn
|
|
8
|
+
Project-URL: Source, https://github.com/OsasDTEch/outturn
|
|
9
|
+
Keywords: voice-agents,llm,evaluation,testing,livekit,agents
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Topic :: Software Development :: Testing
|
|
14
|
+
Requires-Python: >=3.10
|
|
15
|
+
Description-Content-Type: text/markdown
|
|
16
|
+
License-File: LICENSE
|
|
17
|
+
Requires-Dist: PyYAML>=6.0
|
|
18
|
+
Provides-Extra: dev
|
|
19
|
+
Requires-Dist: pytest>=7.4; extra == "dev"
|
|
20
|
+
Provides-Extra: livekit
|
|
21
|
+
Requires-Dist: livekit-agents==1.8.2; extra == "livekit"
|
|
22
|
+
Dynamic: license-file
|
|
23
|
+
|
|
24
|
+
# outturn
|
|
25
|
+
|
|
26
|
+
**Did the agent produce the right outcome? As a number.**
|
|
27
|
+
|
|
28
|
+
A conversation does not end in words. It ends in an order, a booking, a routed ticket, a qualified lead. That is structured data, and structured data can be compared exactly. So whether your agent got it right does not have to be something you learn from a refund. It can be a percentage that runs on every commit.
|
|
29
|
+
|
|
30
|
+
```
|
|
31
|
+
100% simple_booking 8/8
|
|
32
|
+
50% service_switch_duration 4/8 FLAKY
|
|
33
|
+
4x wrong_field: duration_minutes (expected 90, got 60)
|
|
34
|
+
100% never_confirmed 8/8
|
|
35
|
+
|
|
36
|
+
outcome accuracy: 83.3% (20/24 runs)
|
|
37
|
+
flaky scenarios (1): service_switch_duration
|
|
38
|
+
a scenario that passes sometimes is a bug that ships sometimes.
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
---
|
|
42
|
+
|
|
43
|
+
## Why this exists
|
|
44
|
+
|
|
45
|
+
Most voice agent testing is a person calling the number and listening. That finds the bug in front of you and none of the others, it cannot run in CI, and it produces an opinion instead of a number.
|
|
46
|
+
|
|
47
|
+
The alternatives are not much better. Asserting on the transcript is asserting on phrasing, which changes every time you touch the prompt. Using a model to grade the conversation means one nondeterministic system judging another.
|
|
48
|
+
|
|
49
|
+
The outcome is the way out. It is the thing the customer actually receives, it is already structured, and it can be compared exactly.
|
|
50
|
+
|
|
51
|
+
## Three design decisions
|
|
52
|
+
|
|
53
|
+
**Numbers are exact, words are fuzzy.** A quantity of 3 is not close to 2, and a 90 minute appointment is not close to a 60 minute one. But "deep tissue massage" and "Deep Tissue Massage (60min)" are the same service. So every number is compared exactly and every word by normalized similarity.
|
|
54
|
+
|
|
55
|
+
**Every scenario runs N times and reports a pass rate.** Agents are nondeterministic. One green run is not evidence. A bug that appears one run in five is invisible to a boolean test and obvious in a percentage, so outturn reports rates and flags anything that passed sometimes and failed sometimes as FLAKY. Intermittent is worse than broken, because intermittent hides.
|
|
56
|
+
|
|
57
|
+
**An outcome is repeated things plus scalars.** That is all. An order is entries plus a total. A booking is no entries at all, just fields. A triage is fields. One model, no special cases per domain, which is why the same engine handles an ordering agent and a scheduling agent without a line of new code.
|
|
58
|
+
|
|
59
|
+
## Install
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
pip install outturn
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
Or from source, if you want the demo scenarios to play with:
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
git clone https://github.com/OsasDTEch/outturn
|
|
69
|
+
cd outturn
|
|
70
|
+
pip install -e ".[dev]"
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
## Run it right now
|
|
74
|
+
|
|
75
|
+
Two deliberately imperfect demo agents ship with the repo, so you can see real output before writing any integration.
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
outturn scenarios/order --adapter demo-order --runs 8 --seed 42
|
|
79
|
+
outturn scenarios/booking --adapter demo-booking --runs 8 --seed 42
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
Each fails on purpose, because an eval you cannot fail is not an eval.
|
|
83
|
+
|
|
84
|
+
The ordering agent drops a modifier at random under load, which is what FLAKY looks like, and its scope guard is tuned too tight so "do you sell garlic bread" is treated as an off topic question rather than a customer trying to order. That second one is not a strawman, it is behaviour observed on a shipped restaurant agent.
|
|
85
|
+
|
|
86
|
+
The booking agent forgets to carry the duration across when the customer switches service mid call. A 90 minute massage in a 60 minute slot double books the therapist, and nobody finds out until the day.
|
|
87
|
+
|
|
88
|
+
## Writing a scenario
|
|
89
|
+
|
|
90
|
+
Turns in, expected outcome out.
|
|
91
|
+
|
|
92
|
+
**An order**, which has line items:
|
|
93
|
+
|
|
94
|
+
```yaml
|
|
95
|
+
id: modifier_stacking
|
|
96
|
+
description: >
|
|
97
|
+
Two modifiers on one item, added in a separate turn from the item itself.
|
|
98
|
+
tags: [order, modifiers]
|
|
99
|
+
|
|
100
|
+
turns:
|
|
101
|
+
- "Hi, can I get two large pepperoni pizzas"
|
|
102
|
+
- "Actually make those thin crust"
|
|
103
|
+
- "And extra cheese on them please"
|
|
104
|
+
|
|
105
|
+
expect:
|
|
106
|
+
entries:
|
|
107
|
+
- name: pepperoni pizza
|
|
108
|
+
quantity: 2
|
|
109
|
+
unit_price: 12.00
|
|
110
|
+
tags: [thin crust, extra cheese]
|
|
111
|
+
total: 24.00
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
**A booking**, which has none:
|
|
115
|
+
|
|
116
|
+
```yaml
|
|
117
|
+
id: service_switch_duration
|
|
118
|
+
turns:
|
|
119
|
+
- "Can I book a deep tissue massage for Friday"
|
|
120
|
+
- "Actually make it a sports massage instead"
|
|
121
|
+
- "4:00 pm, and yes please book it"
|
|
122
|
+
|
|
123
|
+
expect:
|
|
124
|
+
service: sports massage
|
|
125
|
+
day: friday
|
|
126
|
+
time: "16:00"
|
|
127
|
+
duration_minutes: 90
|
|
128
|
+
confirmed: true
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
Same engine. Anything numeric under an entry becomes an exact comparison. Anything at the top level is a field: numbers and booleans exact, strings fuzzy.
|
|
132
|
+
|
|
133
|
+
You can also assert on a reply, though use it sparingly since phrasing is the brittle part:
|
|
134
|
+
|
|
135
|
+
```yaml
|
|
136
|
+
turns:
|
|
137
|
+
- user: "Do you sell garlic bread?"
|
|
138
|
+
expect_reply_contains: ["garlic bread"]
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
## Connecting your own agent
|
|
142
|
+
|
|
143
|
+
Implement three methods.
|
|
144
|
+
|
|
145
|
+
```python
|
|
146
|
+
from outturn.models import Outcome, Entry
|
|
147
|
+
|
|
148
|
+
class MyAgent:
|
|
149
|
+
name = "my-agent"
|
|
150
|
+
|
|
151
|
+
def reset(self) -> None:
|
|
152
|
+
"""Fresh conversation. Called before every run."""
|
|
153
|
+
self.session = start_session()
|
|
154
|
+
|
|
155
|
+
def send(self, text: str) -> str:
|
|
156
|
+
"""One caller turn in, the agent's reply out."""
|
|
157
|
+
return self.session.turn(text)
|
|
158
|
+
|
|
159
|
+
def outcome(self) -> Outcome:
|
|
160
|
+
"""The structured result, right now."""
|
|
161
|
+
return Outcome(
|
|
162
|
+
entries=tuple(
|
|
163
|
+
Entry(name=l.name,
|
|
164
|
+
numbers={"quantity": l.qty, "unit_price": l.price},
|
|
165
|
+
tags=tuple(l.modifiers))
|
|
166
|
+
for l in self.session.order.lines
|
|
167
|
+
),
|
|
168
|
+
fields={"total": self.session.order.total},
|
|
169
|
+
)
|
|
170
|
+
```
|
|
171
|
+
|
|
172
|
+
Then point outturn at it. **Your agent lives in your repo, not in this one.**
|
|
173
|
+
|
|
174
|
+
```bash
|
|
175
|
+
outturn scenarios/ --adapter myproject.agents:MyAgent --runs 10
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
Anything importable works. If the object is missing one of the three methods,
|
|
179
|
+
outturn says so before the run starts rather than failing with an
|
|
180
|
+
AttributeError halfway through.
|
|
181
|
+
|
|
182
|
+
If your agent cannot hand back a structured outcome, outturn cannot help you, and that is itself the finding. State that lives only in the model's context is not inspectable, and what is not inspectable is not testable. Move the outcome into code and the tool works.
|
|
183
|
+
|
|
184
|
+
## LiveKit
|
|
185
|
+
|
|
186
|
+
There is a driver at `outturn/adapters/livekit.py`. It takes a session factory and one callable that reads your state and returns an Outcome, because the outcome does not live in LiveKit, it lives in whatever your own function tools wrote it to.
|
|
187
|
+
|
|
188
|
+
**Verified against livekit-agents 1.8.2.** Tested with the ollama.com cloud API (model `gemma4:31b`, OpenAI-compatible endpoint). Confirmed:
|
|
189
|
+
|
|
190
|
+
- A session starts with no room and no audio.
|
|
191
|
+
- State written by function tools persists across multiple `session.run()` calls on the same session, so multi-turn scenarios work as written.
|
|
192
|
+
- The assistant reply is a `ChatMessageEvent` in `result.events` with `item.content[0]` as the text.
|
|
193
|
+
|
|
194
|
+
Pin your own install to `livekit-agents==1.8.2` until you have tested against a newer version. This API is young and the `RunResult` shape has changed between releases.
|
|
195
|
+
|
|
196
|
+
## In CI
|
|
197
|
+
|
|
198
|
+
```bash
|
|
199
|
+
outturn scenarios/booking --adapter demo-booking --runs 10 --threshold 0.95
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
Exits non zero if overall accuracy falls below the threshold. Every bug you fix becomes a scenario, so it can never come back silently. `--json` gives machine readable output for tracking accuracy over time.
|
|
203
|
+
|
|
204
|
+
## The scenarios worth writing
|
|
205
|
+
|
|
206
|
+
The ones that break agents, roughly in order of how often they do:
|
|
207
|
+
|
|
208
|
+
- **Corrections mid call.** The customer changes their mind after the agent has already recorded something. This is where most outcomes go wrong, and the failure is usually a field that did not get updated alongside the one that did.
|
|
209
|
+
- **Reference without naming.** "Make it three" with no item named.
|
|
210
|
+
- **Cancellation.** Added, then removed. Do the derived values follow?
|
|
211
|
+
- **Confirmation.** Did the agent act on intent rather than on an actual yes? Booking a customer who was still thinking is a real and expensive failure.
|
|
212
|
+
- **Name collisions.** Two items or two services that sound alike over a phone line.
|
|
213
|
+
- **Illegal combinations.** Extra cheese on a coke. Should be rejected by schema, not accepted politely and discovered later.
|
|
214
|
+
- **Out of scope questions that are really orders.** "Do you sell X" is a customer trying to buy X.
|
|
215
|
+
|
|
216
|
+
## What this does not test
|
|
217
|
+
|
|
218
|
+
outturn drives your agent through text. That is fast enough to run in CI on every commit, and it isolates the reasoning and outcome layer from the audio layer.
|
|
219
|
+
|
|
220
|
+
It does not test speech to text, endpointing or turn taking, and those are real sources of failure, particularly on telephony where audio is narrowband and degrades worst on exactly what matters here: names, numbers and proper nouns.
|
|
221
|
+
|
|
222
|
+
**So this measures whether your agent understands correctly, not whether it hears correctly.** Both matter. For the timing half of the picture, see [voice-latency-profiler](https://github.com/OsasDTEch/voice-latency-profiler).
|
|
223
|
+
|
|
224
|
+
## Status
|
|
225
|
+
|
|
226
|
+
v0.1. The core works and is tested, 31 tests. Roadmap, roughly in order:
|
|
227
|
+
|
|
228
|
+
- [x] LiveKit driver verified against livekit-agents 1.8.2 (ollama.com cloud, `gemma4:31b`)
|
|
229
|
+
- [ ] Path assertions: which tools were called, with which arguments
|
|
230
|
+
- [ ] Pipecat driver
|
|
231
|
+
- [ ] Audio mode, TTS in and STT out, for end to end runs
|
|
232
|
+
- [ ] Accuracy tracked over time, so regressions show as a trend
|
|
233
|
+
|
|
234
|
+
`PRD.md` and `TRD.md` in this repo cover the reasoning and the internals.
|
|
235
|
+
|
|
236
|
+
This project supersedes [voice-agent-evals](https://github.com/OsasDTEch/voice-agent-evals), which asserted on carts only. An order is one kind of outcome among many, and the narrower version was useful to about a tenth of the agents worth testing.
|
|
237
|
+
|
|
238
|
+
## Licence
|
|
239
|
+
|
|
240
|
+
MIT. Built by [Wisdom Omons](https://linkedin.com/in/omons-wisdom).
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
outturn/__init__.py,sha256=7tECRjTfloJxUAFQMhiyl8Oj3DotJuN9z3lQOrWJFn8,251
|
|
2
|
+
outturn/__main__.py,sha256=E6Gls0DNz8GQK2K-kOUIx8cYhgANW_CH54VKrfCfs14,52
|
|
3
|
+
outturn/assertions.py,sha256=tTx1Sv3S1xxfKUOTZe02fIN38NQhSM_9iqVKTAzUP8E,6243
|
|
4
|
+
outturn/cli.py,sha256=127Hwi3too_Rz1h1DlWKXuU3YYwufZFfGz2C9D2jMOI,2115
|
|
5
|
+
outturn/models.py,sha256=o6a-_z8YCtE4iL5-hoAru0X2toPuft8L_3-Bow-6Ct0,4564
|
|
6
|
+
outturn/report.py,sha256=Ev4in-SAjtPQIPrRsLZxBydyoLhljXwyXiirpm1MeIw,2391
|
|
7
|
+
outturn/runner.py,sha256=RACFiJ4ZD2G0GSCpZd4FayA8pyqGIqM1iUlOB6fzxqs,1744
|
|
8
|
+
outturn/scenario.py,sha256=XmJ5GQ4DzUYmcTV3el4NXYwjAE9CypzFxGcxIWMICH8,2446
|
|
9
|
+
outturn/adapters/__init__.py,sha256=F4wvsQAcf-fgERHkQXYmXvhdqnjnBlqAiY-Ln7pty-M,2114
|
|
10
|
+
outturn/adapters/base.py,sha256=Xsk0ymoZbr7donfiTIzpEgq3bM7zI9-TC8trbtXgllU,755
|
|
11
|
+
outturn/adapters/demo_booking.py,sha256=PCkjsFxab9yoODGHU78Ua87xu0sgC6wxveANP6AXcCQ,2622
|
|
12
|
+
outturn/adapters/demo_order.py,sha256=w1UqGXfRzshCyBL0Vv11qYqkKmly0cxZWaDA-x11ICQ,3990
|
|
13
|
+
outturn/adapters/livekit.py,sha256=eA1_NTvbpzQoVVmzTfu64H9ZtfqsAglf21s4LOzG6Rg,5815
|
|
14
|
+
outturn-0.1.0.dist-info/licenses/LICENSE,sha256=kM3jvrV9E_rQdz7LH8CfBIuunYVt5FIXd2kiVoRQu08,1069
|
|
15
|
+
outturn-0.1.0.dist-info/METADATA,sha256=D1a-Jhq33C_MhXE9pOVEzvXKwE7fgj-aRzYyU_zVWqU,10673
|
|
16
|
+
outturn-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
17
|
+
outturn-0.1.0.dist-info/entry_points.txt,sha256=Q1mlhUgk2gC4AlyKx2A2JJGi5nuZd5uZctaGgM1p2E8,45
|
|
18
|
+
outturn-0.1.0.dist-info/top_level.txt,sha256=2kGQyWMtSL4ys27HOT8DpQ28PZg_VqKC-W6s96hz9bo,8
|
|
19
|
+
outturn-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Wisdom Omons
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
outturn
|