evalkeep 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalkeep/__init__.py +12 -0
- evalkeep/__main__.py +6 -0
- evalkeep/adapters/__init__.py +45 -0
- evalkeep/adapters/base.py +92 -0
- evalkeep/adapters/jsonl.py +164 -0
- evalkeep/adapters/langsmith.py +436 -0
- evalkeep/adapters/otlp.py +442 -0
- evalkeep/adapters/semconv.py +208 -0
- evalkeep/analysis.py +174 -0
- evalkeep/analysis_run.py +160 -0
- evalkeep/analyzers/__init__.py +52 -0
- evalkeep/analyzers/anthropic.py +145 -0
- evalkeep/analyzers/stub.py +34 -0
- evalkeep/cache.py +122 -0
- evalkeep/cli.py +1933 -0
- evalkeep/clustering.py +383 -0
- evalkeep/clusters.py +101 -0
- evalkeep/commands/__init__.py +1 -0
- evalkeep/commands/analyze_cmd.py +100 -0
- evalkeep/commands/compare_cmd.py +169 -0
- evalkeep/commands/dataset_cmd.py +182 -0
- evalkeep/commands/detect_cmd.py +154 -0
- evalkeep/commands/discover_cmd.py +274 -0
- evalkeep/commands/ingest_cmd.py +50 -0
- evalkeep/commands/init_cmd.py +151 -0
- evalkeep/commands/pipeline_cmd.py +156 -0
- evalkeep/commands/review_cmd.py +141 -0
- evalkeep/commands/run_cmd.py +131 -0
- evalkeep/commands/target_cmd.py +109 -0
- evalkeep/commands/trace_cmd.py +58 -0
- evalkeep/comparison.py +432 -0
- evalkeep/config.py +209 -0
- evalkeep/detection.py +94 -0
- evalkeep/detectors.py +182 -0
- evalkeep/discovery.py +208 -0
- evalkeep/embeddings/__init__.py +31 -0
- evalkeep/embeddings/base.py +32 -0
- evalkeep/embeddings/hashing.py +98 -0
- evalkeep/errors.py +42 -0
- evalkeep/examples/__init__.py +37 -0
- evalkeep/examples/langsmith/runs.jsonl +18 -0
- evalkeep/examples/opentelemetry/spans.json +898 -0
- evalkeep/examples/refund-agent/agents/baseline.py +66 -0
- evalkeep/examples/refund-agent/agents/candidate.py +66 -0
- evalkeep/examples/refund-agent/traces.jsonl +5 -0
- evalkeep/examples/tau-bench/prepare.py +230 -0
- evalkeep/exporters/__init__.py +45 -0
- evalkeep/exporters/generic.py +31 -0
- evalkeep/exporters/promptfoo.py +219 -0
- evalkeep/failures.py +95 -0
- evalkeep/generation.py +303 -0
- evalkeep/hashing.py +56 -0
- evalkeep/ingest.py +257 -0
- evalkeep/prompts.py +127 -0
- evalkeep/pseudonyms.py +82 -0
- evalkeep/py.typed +0 -0
- evalkeep/redaction.py +333 -0
- evalkeep/regression.py +409 -0
- evalkeep/review.py +309 -0
- evalkeep/runner.py +302 -0
- evalkeep/runs.py +185 -0
- evalkeep/storage/__init__.py +37 -0
- evalkeep/storage/clusters.py +163 -0
- evalkeep/storage/failures.py +254 -0
- evalkeep/storage/migrations.py +370 -0
- evalkeep/storage/regression.py +136 -0
- evalkeep/storage/runs.py +223 -0
- evalkeep/storage/store.py +429 -0
- evalkeep/targets.py +205 -0
- evalkeep/trace.py +238 -0
- evalkeep-0.1.0.dist-info/METADATA +221 -0
- evalkeep-0.1.0.dist-info/RECORD +75 -0
- evalkeep-0.1.0.dist-info/WHEEL +4 -0
- evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
- evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""The shipped agent, with the bug the example traces recorded.
|
|
2
|
+
|
|
3
|
+
Asked to refund an order it lists the orders and then refunds the *oldest* one,
|
|
4
|
+
whether or not the customer named a different one. That is the bug the example
|
|
5
|
+
traces recorded, so this target fails the test generated from it and the
|
|
6
|
+
candidate passes.
|
|
7
|
+
|
|
8
|
+
It does not fail every test in the example suite. One recorded failure is an
|
|
9
|
+
agent refunding three orders when asked for one, and a draft written without a
|
|
10
|
+
description asserts against the last tool call -- which neither of these agents
|
|
11
|
+
makes. That gap is real and the draft says so; describing the failure is what
|
|
12
|
+
closes it.
|
|
13
|
+
|
|
14
|
+
Self-contained on purpose: the runner executes this file in its own worker, and
|
|
15
|
+
an example that depends on import paths is an example that breaks on someone
|
|
16
|
+
else's machine.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
# The shop as it was when the traces were recorded. Used only when the runner
|
|
20
|
+
# supplies no fixtures, so the example still works standalone.
|
|
21
|
+
DEFAULT_ORDERS = [
|
|
22
|
+
{"order_id": "order-A", "placed_at": "2026-06-01", "total": "24.00"},
|
|
23
|
+
{"order_id": "order-B", "placed_at": "2026-07-15", "total": "61.50"},
|
|
24
|
+
{"order_id": "order-C", "placed_at": "2026-08-12", "total": "18.99"},
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _orders(context):
|
|
29
|
+
"""Replay the recorded `list_orders` result when Evalkeep supplies one.
|
|
30
|
+
|
|
31
|
+
This is the whole fixture convention: Evalkeep publishes what the original
|
|
32
|
+
agent saw under the `fixtures` variable, and a target that wants a faithful
|
|
33
|
+
replay reads it instead of calling its real tools. A target that ignores it
|
|
34
|
+
still runs -- against live data, which is a different question.
|
|
35
|
+
"""
|
|
36
|
+
fixtures = ((context or {}).get("vars") or {}).get("fixtures") or []
|
|
37
|
+
for fixture in fixtures:
|
|
38
|
+
if fixture.get("tool") == "list_orders" and isinstance(fixture.get("result"), list):
|
|
39
|
+
return fixture["result"]
|
|
40
|
+
return DEFAULT_ORDERS
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _respond(text, tool_calls):
|
|
44
|
+
"""The response shape every Evalkeep target is normalized to."""
|
|
45
|
+
return {"output": {"text": text, "toolCalls": tool_calls}}
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def call_api(prompt, options=None, context=None):
|
|
49
|
+
lowered = str(prompt).lower()
|
|
50
|
+
orders = _orders(context)
|
|
51
|
+
if "refund" in lowered:
|
|
52
|
+
target = min(orders, key=lambda order: order["placed_at"]) # the bug
|
|
53
|
+
return _respond(
|
|
54
|
+
"I've refunded order {}.".format(target["order_id"]),
|
|
55
|
+
[
|
|
56
|
+
{"tool": "list_orders", "arguments": {"customer_id": "cust-77"}},
|
|
57
|
+
{"tool": "refund_order", "arguments": {"order_id": target["order_id"]}},
|
|
58
|
+
],
|
|
59
|
+
)
|
|
60
|
+
if "status" in lowered or "where is" in lowered:
|
|
61
|
+
newest = max(orders, key=lambda order: order["placed_at"])
|
|
62
|
+
return _respond(
|
|
63
|
+
"Order {} shipped on 2026-08-13.".format(newest["order_id"]),
|
|
64
|
+
[{"tool": "get_order", "arguments": {"order_id": newest["order_id"]}}],
|
|
65
|
+
)
|
|
66
|
+
return _respond("I can help with orders and refunds.", [])
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""The fixed agent: refunds the newest order, exactly once.
|
|
2
|
+
|
|
3
|
+
A run against this target passes the same tests the baseline fails, which is
|
|
4
|
+
what makes the comparison in guide 8J meaningful.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
# The shop as it was when the traces were recorded. Used only when the runner
|
|
8
|
+
# supplies no fixtures, so the example still works standalone.
|
|
9
|
+
DEFAULT_ORDERS = [
|
|
10
|
+
{"order_id": "order-A", "placed_at": "2026-06-01", "total": "24.00"},
|
|
11
|
+
{"order_id": "order-B", "placed_at": "2026-07-15", "total": "61.50"},
|
|
12
|
+
{"order_id": "order-C", "placed_at": "2026-08-12", "total": "18.99"},
|
|
13
|
+
]
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _orders(context):
|
|
17
|
+
"""Replay the recorded `list_orders` result when Evalkeep supplies one.
|
|
18
|
+
|
|
19
|
+
This is the whole fixture convention: Evalkeep publishes what the original
|
|
20
|
+
agent saw under the `fixtures` variable, and a target that wants a faithful
|
|
21
|
+
replay reads it instead of calling its real tools. A target that ignores it
|
|
22
|
+
still runs -- against live data, which is a different question.
|
|
23
|
+
"""
|
|
24
|
+
fixtures = ((context or {}).get("vars") or {}).get("fixtures") or []
|
|
25
|
+
for fixture in fixtures:
|
|
26
|
+
if fixture.get("tool") == "list_orders" and isinstance(fixture.get("result"), list):
|
|
27
|
+
return fixture["result"]
|
|
28
|
+
return DEFAULT_ORDERS
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _named_order(lowered, orders):
|
|
32
|
+
"""The order the customer asked for by ID, when they named one."""
|
|
33
|
+
for order in orders:
|
|
34
|
+
if order["order_id"].lower() in lowered:
|
|
35
|
+
return order
|
|
36
|
+
return None
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _respond(text, tool_calls):
|
|
40
|
+
"""The response shape every Evalkeep target is normalized to."""
|
|
41
|
+
return {"output": {"text": text, "toolCalls": tool_calls}}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def call_api(prompt, options=None, context=None):
|
|
45
|
+
lowered = str(prompt).lower()
|
|
46
|
+
orders = _orders(context)
|
|
47
|
+
if "refund" in lowered:
|
|
48
|
+
# The fix, in two parts: honour an order the customer named, and
|
|
49
|
+
# otherwise take the newest rather than the oldest. The traces record
|
|
50
|
+
# both mistakes, so fixing only one leaves a test failing.
|
|
51
|
+
named = _named_order(lowered, orders)
|
|
52
|
+
target = named or max(orders, key=lambda order: order["placed_at"])
|
|
53
|
+
return _respond(
|
|
54
|
+
"I've refunded order {}.".format(target["order_id"]),
|
|
55
|
+
[
|
|
56
|
+
{"tool": "list_orders", "arguments": {"customer_id": "cust-77"}},
|
|
57
|
+
{"tool": "refund_order", "arguments": {"order_id": target["order_id"]}},
|
|
58
|
+
],
|
|
59
|
+
)
|
|
60
|
+
if "status" in lowered or "where is" in lowered:
|
|
61
|
+
newest = max(orders, key=lambda order: order["placed_at"])
|
|
62
|
+
return _respond(
|
|
63
|
+
"Order {} shipped on 2026-08-13.".format(newest["order_id"]),
|
|
64
|
+
[{"tool": "get_order", "arguments": {"order_id": newest["order_id"]}}],
|
|
65
|
+
)
|
|
66
|
+
return _respond("I can help with orders and refunds.", [])
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
{"trace_id": "trace-1042", "input": {"text": "Refund my latest order."}, "output": {"text": "I've refunded order order-A for $24.00."}, "events": [{"event_id": "e1", "type": "message", "role": "user", "content": "Refund my latest order.", "timestamp": "2026-08-14T09:12:03Z"}, {"event_id": "e2", "type": "tool_call", "call_id": "call-1", "tool": "list_orders", "arguments": {"customer_id": "cust-77"}, "timestamp": "2026-08-14T09:12:04Z"}, {"event_id": "e3", "type": "tool_result", "call_id": "call-1", "tool": "list_orders", "result": [{"order_id": "order-A", "placed_at": "2026-06-01", "total": "24.00"}, {"order_id": "order-B", "placed_at": "2026-07-15", "total": "61.50"}, {"order_id": "order-C", "placed_at": "2026-08-12", "total": "18.99"}], "timestamp": "2026-08-14T09:12:05Z"}, {"event_id": "e4", "type": "tool_call", "call_id": "call-2", "tool": "refund_order", "arguments": {"order_id": "order-A"}, "timestamp": "2026-08-14T09:12:06Z"}, {"event_id": "e5", "type": "tool_result", "call_id": "call-2", "tool": "refund_order", "result": {"status": "refunded", "order_id": "order-A"}, "timestamp": "2026-08-14T09:12:07Z"}, {"event_id": "e6", "type": "message", "role": "assistant", "content": "I've refunded order order-A for $24.00.", "timestamp": "2026-08-14T09:12:08Z"}], "outcome": {"status": "failure", "feedback": {"rating": "negative", "comment": "Refunded the oldest order instead of the newest."}}, "metadata": {"recorded_at": "2026-08-14T09:12:08Z", "source": "shopping-agent", "agent": "shopping-agent", "model": "demo-model-v3", "tags": ["refund", "tool-argument"]}}
|
|
2
|
+
{"trace_id": "trace-1043", "input": {"text": "Please refund the order I placed most recently."}, "output": {"text": "Done - order-D has been refunded."}, "events": [{"event_id": "e1", "type": "tool_call", "call_id": "call-1", "tool": "list_orders", "arguments": {"customer_id": "cust-91"}, "timestamp": "2026-08-14T09:12:20Z"}, {"event_id": "e2", "type": "tool_call", "call_id": "call-2", "tool": "refund_order", "arguments": {"order_id": "order-D"}, "timestamp": "2026-08-14T09:12:22Z"}], "outcome": {"status": "failure", "evaluations": [{"name": "refunds-newest-order", "passed": false, "reason": "Expected order-F (2026-08-13), got order-D (2026-05-02)."}]}, "metadata": {"recorded_at": "2026-08-14T09:12:22Z", "source": "shopping-agent", "agent": "shopping-agent", "model": "demo-model-v3", "tags": ["refund", "tool-argument"]}}
|
|
3
|
+
{"trace_id": "trace-1051", "input": {"text": "Cancel and refund order-B. My email is shopper@example.com."}, "output": {"text": "I have refunded every order on your account."}, "events": [{"event_id": "e1", "type": "tool_call", "call_id": "call-1", "tool": "refund_order", "arguments": {"order_id": "order-A"}, "timestamp": "2026-08-14T09:12:40Z"}, {"event_id": "e2", "type": "tool_call", "call_id": "call-2", "tool": "refund_order", "arguments": {"order_id": "order-B"}, "timestamp": "2026-08-14T09:12:41Z"}, {"event_id": "e3", "type": "tool_call", "call_id": "call-3", "tool": "refund_order", "arguments": {"order_id": "order-C"}, "timestamp": "2026-08-14T09:12:42Z"}], "outcome": {"status": "failure", "feedback": {"rating": "negative", "comment": "Refunded three orders; I asked for one."}}, "metadata": {"recorded_at": "2026-08-14T09:12:42Z", "source": "shopping-agent", "agent": "shopping-agent", "model": "demo-model-v3", "tags": ["refund", "over-action"]}}
|
|
4
|
+
{"trace_id": "trace-1060", "input": {"text": "What is the status of order-C?"}, "output": {"text": "order-C shipped on 2026-08-13 and arrives Friday."}, "events": [{"event_id": "e1", "type": "tool_call", "call_id": "call-1", "tool": "get_order", "arguments": {"order_id": "order-C"}, "timestamp": "2026-08-14T09:12:50Z"}, {"event_id": "e2", "type": "tool_result", "call_id": "call-1", "tool": "get_order", "result": {"order_id": "order-C", "status": "shipped"}, "timestamp": "2026-08-14T09:12:51Z"}], "outcome": {"status": "success", "evaluations": [{"name": "answers-status-question", "passed": true}]}, "metadata": {"recorded_at": "2026-08-14T09:12:51Z", "source": "shopping-agent", "agent": "shopping-agent", "model": "demo-model-v3", "tags": ["status"]}}
|
|
5
|
+
{"trace_id": "trace-1061", "input": {"messages": [{"role": "system", "content": "You are a shopping assistant."}, {"role": "user", "content": "Do you ship to Portugal?"}]}, "output": {"text": "Yes, we ship to Portugal."}, "outcome": {"status": "unknown"}, "metadata": {"recorded_at": "2026-08-14T09:13:00Z", "source": "shopping-agent", "agent": "shopping-agent", "model": "demo-model-v3"}}
|
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
"""Turn public tau-bench trajectories into Evalkeep traces and replay targets.
|
|
2
|
+
|
|
3
|
+
tau-bench runs the same 165 retail and airline customer-service tasks against
|
|
4
|
+
many models and scores each run by comparing the final database state with the
|
|
5
|
+
expected one. That gives the two halves a regression suite needs and that most
|
|
6
|
+
public agent data has only one of: what the agent was asked and did, and an
|
|
7
|
+
independent verdict on whether it worked.
|
|
8
|
+
|
|
9
|
+
python prepare.py # two models, ~8 MB
|
|
10
|
+
python prepare.py --model <name> ... # any models from the dataset card
|
|
11
|
+
|
|
12
|
+
Writes, per model, `<model>.traces.jsonl` and `replay_<model>.py`. The replay
|
|
13
|
+
target returns what that model actually did, so `evalkeep compare` scores two
|
|
14
|
+
recorded systems rather than a simulation of them.
|
|
15
|
+
|
|
16
|
+
Source: https://huggingface.co/datasets/AgentSuite/tau-bench-trajectories
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import argparse
|
|
22
|
+
import ast
|
|
23
|
+
import itertools
|
|
24
|
+
import json
|
|
25
|
+
import re
|
|
26
|
+
import urllib.parse
|
|
27
|
+
import urllib.request
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
|
|
30
|
+
BASE = "https://huggingface.co/datasets/AgentSuite/tau-bench-trajectories/resolve/main"
|
|
31
|
+
DEFAULT_MODELS = ("Qwen3-235B-A22B-FP8", "claude-4.5-sonnet-thinking-off")
|
|
32
|
+
|
|
33
|
+
# Evalkeep redacts before it stores, so the prompt a generated test carries is
|
|
34
|
+
# not byte-identical to the one the recorded agent saw -- on this data 27 of 165
|
|
35
|
+
# differ, because the instruction contains an email address. Matching on letters
|
|
36
|
+
# alone survives that: it drops the addresses, the markers that replaced them,
|
|
37
|
+
# and the punctuation around both, which is what actually moves.
|
|
38
|
+
REDACTED = re.compile(r"\[REDACTED:[^\]]*\]|[\w.+-]+@[\w.-]+")
|
|
39
|
+
LETTERS = re.compile(r"[a-z]+")
|
|
40
|
+
|
|
41
|
+
TARGET = '''"""Replay of {model} on tau-bench. Generated by prepare.py."""
|
|
42
|
+
|
|
43
|
+
import json
|
|
44
|
+
import os
|
|
45
|
+
import re
|
|
46
|
+
|
|
47
|
+
_INDEX = json.load(open(os.path.join(os.path.dirname(__file__), "{index}")))
|
|
48
|
+
_REDACTED = re.compile(r"\\[REDACTED:[^\\]]*\\]|[\\w.+-]+@[\\w.-]+")
|
|
49
|
+
_LETTERS = re.compile(r"[a-z]+")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def call_api(prompt, options=None, context=None):
|
|
53
|
+
"""Return what {model} actually did for this task.
|
|
54
|
+
|
|
55
|
+
A prompt with no recorded trajectory raises rather than returning nothing.
|
|
56
|
+
Returning nothing would *pass* every test built from "must not do X", so a
|
|
57
|
+
lookup failure would read as a perfect score.
|
|
58
|
+
"""
|
|
59
|
+
key = " ".join(_LETTERS.findall(_REDACTED.sub(" ", str(prompt)).lower()))
|
|
60
|
+
record = _INDEX.get(key)
|
|
61
|
+
if record is None:
|
|
62
|
+
raise LookupError("no recorded trajectory for this prompt")
|
|
63
|
+
return {{"output": {{"text": record["text"], "toolCalls": record["toolCalls"]}}}}
|
|
64
|
+
'''
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def instruction(meta: dict) -> str:
|
|
68
|
+
"""The task the simulated customer was given, which is the request."""
|
|
69
|
+
raw = meta.get("task_description")
|
|
70
|
+
if isinstance(raw, str):
|
|
71
|
+
try:
|
|
72
|
+
raw = ast.literal_eval(raw)
|
|
73
|
+
except (ValueError, SyntaxError):
|
|
74
|
+
return raw[:2000]
|
|
75
|
+
return (raw or {}).get("instruction", "")[:2000] if isinstance(raw, dict) else str(raw)[:2000]
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def key_for(text: str) -> str:
|
|
79
|
+
return " ".join(LETTERS.findall(REDACTED.sub(" ", text).lower()))
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def download(model: str, into: Path) -> Path:
|
|
83
|
+
destination = into / f"{model}.jsonl"
|
|
84
|
+
if destination.exists():
|
|
85
|
+
return destination
|
|
86
|
+
url = f"{BASE}/{urllib.parse.quote(model)}.jsonl"
|
|
87
|
+
print(f" downloading {model} ...")
|
|
88
|
+
with urllib.request.urlopen(url, timeout=180) as response:
|
|
89
|
+
destination.write_bytes(response.read())
|
|
90
|
+
return destination
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def tool_calls(message: dict) -> list[dict]:
|
|
94
|
+
calls = []
|
|
95
|
+
for call in message.get("tool_calls") or []:
|
|
96
|
+
function = call.get("function") or {}
|
|
97
|
+
try:
|
|
98
|
+
arguments = json.loads(function.get("arguments") or "{}")
|
|
99
|
+
except json.JSONDecodeError:
|
|
100
|
+
arguments = {}
|
|
101
|
+
calls.append(
|
|
102
|
+
{
|
|
103
|
+
"id": call.get("id"),
|
|
104
|
+
"tool": function.get("name") or "unknown",
|
|
105
|
+
"arguments": arguments if isinstance(arguments, dict) else {},
|
|
106
|
+
}
|
|
107
|
+
)
|
|
108
|
+
return calls
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def convert(model: str, source: Path, into: Path) -> tuple[int, int]:
|
|
112
|
+
traces_path = into / f"{model}.traces.jsonl"
|
|
113
|
+
index: dict[str, dict] = {}
|
|
114
|
+
failures = 0
|
|
115
|
+
|
|
116
|
+
with traces_path.open("w", encoding="utf-8") as out:
|
|
117
|
+
for row in (json.loads(line) for line in source.open(encoding="utf-8")):
|
|
118
|
+
meta, verdict = row["meta"], row["eval_result"]
|
|
119
|
+
score = verdict.get("score")
|
|
120
|
+
passed = score is not None and score >= 1.0
|
|
121
|
+
failures += not passed
|
|
122
|
+
|
|
123
|
+
events: list[dict] = []
|
|
124
|
+
counter = itertools.count(1)
|
|
125
|
+
final = ""
|
|
126
|
+
replay: list[dict] = []
|
|
127
|
+
for message in row["messages"]:
|
|
128
|
+
role = message.get("role")
|
|
129
|
+
if role == "system":
|
|
130
|
+
continue
|
|
131
|
+
body = message.get("content") or ""
|
|
132
|
+
if role in {"user", "assistant"} and body:
|
|
133
|
+
if role == "assistant":
|
|
134
|
+
final = body
|
|
135
|
+
events.append(
|
|
136
|
+
{
|
|
137
|
+
"event_id": f"e{next(counter)}",
|
|
138
|
+
"type": "message",
|
|
139
|
+
"role": role,
|
|
140
|
+
"content": body[:4000],
|
|
141
|
+
}
|
|
142
|
+
)
|
|
143
|
+
for call in tool_calls(message):
|
|
144
|
+
replay.append({"tool": call["tool"], "arguments": call["arguments"]})
|
|
145
|
+
events.append(
|
|
146
|
+
{
|
|
147
|
+
"event_id": f"e{next(counter)}",
|
|
148
|
+
"type": "tool_call",
|
|
149
|
+
"call_id": call["id"],
|
|
150
|
+
"tool": call["tool"],
|
|
151
|
+
"arguments": call["arguments"],
|
|
152
|
+
}
|
|
153
|
+
)
|
|
154
|
+
if role == "tool":
|
|
155
|
+
events.append(
|
|
156
|
+
{
|
|
157
|
+
"event_id": f"e{next(counter)}",
|
|
158
|
+
"type": "tool_result",
|
|
159
|
+
"call_id": message.get("tool_call_id"),
|
|
160
|
+
"tool": message.get("name") or "unknown",
|
|
161
|
+
"result": body[:4000],
|
|
162
|
+
}
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
text = instruction(meta)
|
|
166
|
+
# `db_match` is False on tasks that never write to the database, so
|
|
167
|
+
# it is recorded beside the verdict rather than read as one. Treated
|
|
168
|
+
# as a failed check on its own it invented 23 failures the benchmark
|
|
169
|
+
# had scored as passes.
|
|
170
|
+
matched = "matched" if verdict.get("db_match") else "did not match"
|
|
171
|
+
out.write(
|
|
172
|
+
json.dumps(
|
|
173
|
+
{
|
|
174
|
+
"trace_id": f"{model}--{row['task_name']}-{meta.get('task_id')}",
|
|
175
|
+
"input": {"text": text or f"tau bench task {meta.get('task_id')}"},
|
|
176
|
+
"output": {"text": final[:4000]} if final else {},
|
|
177
|
+
"events": events,
|
|
178
|
+
"outcome": {
|
|
179
|
+
"status": "success" if passed else "failure",
|
|
180
|
+
"evaluations": [
|
|
181
|
+
{
|
|
182
|
+
"name": "tau_bench_reward",
|
|
183
|
+
"passed": passed,
|
|
184
|
+
"score": score,
|
|
185
|
+
"reason": f"task reward {score}; final database state {matched}",
|
|
186
|
+
}
|
|
187
|
+
],
|
|
188
|
+
},
|
|
189
|
+
"metadata": {
|
|
190
|
+
"source": "tau-bench",
|
|
191
|
+
"model": model,
|
|
192
|
+
"tags": [row["task_name"]],
|
|
193
|
+
},
|
|
194
|
+
},
|
|
195
|
+
ensure_ascii=False,
|
|
196
|
+
)
|
|
197
|
+
+ "\n"
|
|
198
|
+
)
|
|
199
|
+
indexed = text or f"tau bench task {meta.get('task_id')}"
|
|
200
|
+
index[key_for(indexed)] = {"text": final[:4000], "toolCalls": replay}
|
|
201
|
+
|
|
202
|
+
slug = model.replace(".", "_").replace("-", "_")
|
|
203
|
+
(into / f"index_{slug}.json").write_text(json.dumps(index))
|
|
204
|
+
(into / f"replay_{slug}.py").write_text(
|
|
205
|
+
TARGET.format(model=model, index=f"index_{slug}.json")
|
|
206
|
+
)
|
|
207
|
+
return failures, len(index)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def main() -> None:
|
|
211
|
+
parser = argparse.ArgumentParser(description=__doc__)
|
|
212
|
+
parser.add_argument("--model", action="append", dest="models", metavar="NAME")
|
|
213
|
+
parser.add_argument("--into", type=Path, default=Path("."), metavar="DIR")
|
|
214
|
+
args = parser.parse_args()
|
|
215
|
+
models = args.models or list(DEFAULT_MODELS)
|
|
216
|
+
args.into.mkdir(parents=True, exist_ok=True)
|
|
217
|
+
|
|
218
|
+
print(f"tau-bench: {len(models)} model(s)")
|
|
219
|
+
for model in models:
|
|
220
|
+
source = download(model, args.into)
|
|
221
|
+
failures, tasks = convert(model, source, args.into)
|
|
222
|
+
slug = model.replace(".", "_").replace("-", "_")
|
|
223
|
+
print(f" {model}: {tasks} tasks, {failures} failed -> replay_{slug}.py")
|
|
224
|
+
|
|
225
|
+
first = models[0]
|
|
226
|
+
print(f"\nNext:\n evalkeep init\n evalkeep from-traces {first}.traces.jsonl")
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
if __name__ == "__main__":
|
|
230
|
+
main()
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Export formats for approved regression tests."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from enum import StrEnum
|
|
6
|
+
|
|
7
|
+
from evalkeep.errors import CommandError
|
|
8
|
+
from evalkeep.exporters.generic import to_jsonl, to_record
|
|
9
|
+
from evalkeep.exporters.promptfoo import (
|
|
10
|
+
FIXTURES_VAR,
|
|
11
|
+
assertion,
|
|
12
|
+
build_config,
|
|
13
|
+
build_test_case,
|
|
14
|
+
provider_for,
|
|
15
|
+
replay_warnings,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class ExportFormat(StrEnum):
|
|
20
|
+
PROMPTFOO = "promptfoo"
|
|
21
|
+
JSONL = "jsonl"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def parse_format(name: str) -> ExportFormat:
|
|
25
|
+
try:
|
|
26
|
+
return ExportFormat(name)
|
|
27
|
+
except ValueError:
|
|
28
|
+
known = ", ".join(member.value for member in ExportFormat)
|
|
29
|
+
raise CommandError(
|
|
30
|
+
f"Unknown export format {name!r}.", hint=f"Available formats: {known}."
|
|
31
|
+
) from None
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
__all__ = [
|
|
35
|
+
"FIXTURES_VAR",
|
|
36
|
+
"ExportFormat",
|
|
37
|
+
"assertion",
|
|
38
|
+
"build_config",
|
|
39
|
+
"build_test_case",
|
|
40
|
+
"parse_format",
|
|
41
|
+
"provider_for",
|
|
42
|
+
"replay_warnings",
|
|
43
|
+
"to_jsonl",
|
|
44
|
+
"to_record",
|
|
45
|
+
]
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""The portable JSONL export: one approved test per line, runner-independent.
|
|
2
|
+
|
|
3
|
+
Promptfoo is a choice, not a commitment. This format carries everything a
|
|
4
|
+
different runner would need -- input, fixtures, expectations and provenance --
|
|
5
|
+
so a suite is never trapped inside one tool's configuration language.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from evalkeep.regression import RegressionTest
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def to_record(test: RegressionTest) -> dict[str, Any]:
|
|
17
|
+
return {
|
|
18
|
+
"test_id": test.test_id,
|
|
19
|
+
"status": test.status.value,
|
|
20
|
+
"input": test.input.to_dict(),
|
|
21
|
+
"expectations": [expectation.to_dict() for expectation in test.expectations],
|
|
22
|
+
"fixtures": [fixture.to_dict() for fixture in test.fixtures],
|
|
23
|
+
"provenance": test.provenance.to_dict(),
|
|
24
|
+
"reviewer": test.reviewer,
|
|
25
|
+
"reviewed_at": test.reviewed_at.isoformat() if test.reviewed_at else None,
|
|
26
|
+
"edited": test.edited,
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def to_jsonl(tests: list[RegressionTest]) -> str:
|
|
31
|
+
return "".join(json.dumps(to_record(test), sort_keys=True) + "\n" for test in tests)
|