evalkeep 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. evalkeep/__init__.py +12 -0
  2. evalkeep/__main__.py +6 -0
  3. evalkeep/adapters/__init__.py +45 -0
  4. evalkeep/adapters/base.py +92 -0
  5. evalkeep/adapters/jsonl.py +164 -0
  6. evalkeep/adapters/langsmith.py +436 -0
  7. evalkeep/adapters/otlp.py +442 -0
  8. evalkeep/adapters/semconv.py +208 -0
  9. evalkeep/analysis.py +174 -0
  10. evalkeep/analysis_run.py +160 -0
  11. evalkeep/analyzers/__init__.py +52 -0
  12. evalkeep/analyzers/anthropic.py +145 -0
  13. evalkeep/analyzers/stub.py +34 -0
  14. evalkeep/cache.py +122 -0
  15. evalkeep/cli.py +1933 -0
  16. evalkeep/clustering.py +383 -0
  17. evalkeep/clusters.py +101 -0
  18. evalkeep/commands/__init__.py +1 -0
  19. evalkeep/commands/analyze_cmd.py +100 -0
  20. evalkeep/commands/compare_cmd.py +169 -0
  21. evalkeep/commands/dataset_cmd.py +182 -0
  22. evalkeep/commands/detect_cmd.py +154 -0
  23. evalkeep/commands/discover_cmd.py +274 -0
  24. evalkeep/commands/ingest_cmd.py +50 -0
  25. evalkeep/commands/init_cmd.py +151 -0
  26. evalkeep/commands/pipeline_cmd.py +156 -0
  27. evalkeep/commands/review_cmd.py +141 -0
  28. evalkeep/commands/run_cmd.py +131 -0
  29. evalkeep/commands/target_cmd.py +109 -0
  30. evalkeep/commands/trace_cmd.py +58 -0
  31. evalkeep/comparison.py +432 -0
  32. evalkeep/config.py +209 -0
  33. evalkeep/detection.py +94 -0
  34. evalkeep/detectors.py +182 -0
  35. evalkeep/discovery.py +208 -0
  36. evalkeep/embeddings/__init__.py +31 -0
  37. evalkeep/embeddings/base.py +32 -0
  38. evalkeep/embeddings/hashing.py +98 -0
  39. evalkeep/errors.py +42 -0
  40. evalkeep/examples/__init__.py +37 -0
  41. evalkeep/examples/langsmith/runs.jsonl +18 -0
  42. evalkeep/examples/opentelemetry/spans.json +898 -0
  43. evalkeep/examples/refund-agent/agents/baseline.py +66 -0
  44. evalkeep/examples/refund-agent/agents/candidate.py +66 -0
  45. evalkeep/examples/refund-agent/traces.jsonl +5 -0
  46. evalkeep/examples/tau-bench/prepare.py +230 -0
  47. evalkeep/exporters/__init__.py +45 -0
  48. evalkeep/exporters/generic.py +31 -0
  49. evalkeep/exporters/promptfoo.py +219 -0
  50. evalkeep/failures.py +95 -0
  51. evalkeep/generation.py +303 -0
  52. evalkeep/hashing.py +56 -0
  53. evalkeep/ingest.py +257 -0
  54. evalkeep/prompts.py +127 -0
  55. evalkeep/pseudonyms.py +82 -0
  56. evalkeep/py.typed +0 -0
  57. evalkeep/redaction.py +333 -0
  58. evalkeep/regression.py +409 -0
  59. evalkeep/review.py +309 -0
  60. evalkeep/runner.py +302 -0
  61. evalkeep/runs.py +185 -0
  62. evalkeep/storage/__init__.py +37 -0
  63. evalkeep/storage/clusters.py +163 -0
  64. evalkeep/storage/failures.py +254 -0
  65. evalkeep/storage/migrations.py +370 -0
  66. evalkeep/storage/regression.py +136 -0
  67. evalkeep/storage/runs.py +223 -0
  68. evalkeep/storage/store.py +429 -0
  69. evalkeep/targets.py +205 -0
  70. evalkeep/trace.py +238 -0
  71. evalkeep-0.1.0.dist-info/METADATA +221 -0
  72. evalkeep-0.1.0.dist-info/RECORD +75 -0
  73. evalkeep-0.1.0.dist-info/WHEEL +4 -0
  74. evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
  75. evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
@@ -0,0 +1,66 @@
1
+ """The shipped agent, with the bug the example traces recorded.
2
+
3
+ Asked to refund an order it lists the orders and then refunds the *oldest* one,
4
+ whether or not the customer named a different one. That is the bug the example
5
+ traces recorded, so this target fails the test generated from it and the
6
+ candidate passes.
7
+
8
+ It does not fail every test in the example suite. One recorded failure is an
9
+ agent refunding three orders when asked for one, and a draft written without a
10
+ description asserts against the last tool call -- which neither of these agents
11
+ makes. That gap is real and the draft says so; describing the failure is what
12
+ closes it.
13
+
14
+ Self-contained on purpose: the runner executes this file in its own worker, and
15
+ an example that depends on import paths is an example that breaks on someone
16
+ else's machine.
17
+ """
18
+
19
+ # The shop as it was when the traces were recorded. Used only when the runner
20
+ # supplies no fixtures, so the example still works standalone.
21
+ DEFAULT_ORDERS = [
22
+ {"order_id": "order-A", "placed_at": "2026-06-01", "total": "24.00"},
23
+ {"order_id": "order-B", "placed_at": "2026-07-15", "total": "61.50"},
24
+ {"order_id": "order-C", "placed_at": "2026-08-12", "total": "18.99"},
25
+ ]
26
+
27
+
28
+ def _orders(context):
29
+ """Replay the recorded `list_orders` result when Evalkeep supplies one.
30
+
31
+ This is the whole fixture convention: Evalkeep publishes what the original
32
+ agent saw under the `fixtures` variable, and a target that wants a faithful
33
+ replay reads it instead of calling its real tools. A target that ignores it
34
+ still runs -- against live data, which is a different question.
35
+ """
36
+ fixtures = ((context or {}).get("vars") or {}).get("fixtures") or []
37
+ for fixture in fixtures:
38
+ if fixture.get("tool") == "list_orders" and isinstance(fixture.get("result"), list):
39
+ return fixture["result"]
40
+ return DEFAULT_ORDERS
41
+
42
+
43
+ def _respond(text, tool_calls):
44
+ """The response shape every Evalkeep target is normalized to."""
45
+ return {"output": {"text": text, "toolCalls": tool_calls}}
46
+
47
+
48
+ def call_api(prompt, options=None, context=None):
49
+ lowered = str(prompt).lower()
50
+ orders = _orders(context)
51
+ if "refund" in lowered:
52
+ target = min(orders, key=lambda order: order["placed_at"]) # the bug
53
+ return _respond(
54
+ "I've refunded order {}.".format(target["order_id"]),
55
+ [
56
+ {"tool": "list_orders", "arguments": {"customer_id": "cust-77"}},
57
+ {"tool": "refund_order", "arguments": {"order_id": target["order_id"]}},
58
+ ],
59
+ )
60
+ if "status" in lowered or "where is" in lowered:
61
+ newest = max(orders, key=lambda order: order["placed_at"])
62
+ return _respond(
63
+ "Order {} shipped on 2026-08-13.".format(newest["order_id"]),
64
+ [{"tool": "get_order", "arguments": {"order_id": newest["order_id"]}}],
65
+ )
66
+ return _respond("I can help with orders and refunds.", [])
@@ -0,0 +1,66 @@
1
+ """The fixed agent: refunds the newest order, exactly once.
2
+
3
+ A run against this target passes the same tests the baseline fails, which is
4
+ what makes the comparison in guide 8J meaningful.
5
+ """
6
+
7
+ # The shop as it was when the traces were recorded. Used only when the runner
8
+ # supplies no fixtures, so the example still works standalone.
9
+ DEFAULT_ORDERS = [
10
+ {"order_id": "order-A", "placed_at": "2026-06-01", "total": "24.00"},
11
+ {"order_id": "order-B", "placed_at": "2026-07-15", "total": "61.50"},
12
+ {"order_id": "order-C", "placed_at": "2026-08-12", "total": "18.99"},
13
+ ]
14
+
15
+
16
+ def _orders(context):
17
+ """Replay the recorded `list_orders` result when Evalkeep supplies one.
18
+
19
+ This is the whole fixture convention: Evalkeep publishes what the original
20
+ agent saw under the `fixtures` variable, and a target that wants a faithful
21
+ replay reads it instead of calling its real tools. A target that ignores it
22
+ still runs -- against live data, which is a different question.
23
+ """
24
+ fixtures = ((context or {}).get("vars") or {}).get("fixtures") or []
25
+ for fixture in fixtures:
26
+ if fixture.get("tool") == "list_orders" and isinstance(fixture.get("result"), list):
27
+ return fixture["result"]
28
+ return DEFAULT_ORDERS
29
+
30
+
31
+ def _named_order(lowered, orders):
32
+ """The order the customer asked for by ID, when they named one."""
33
+ for order in orders:
34
+ if order["order_id"].lower() in lowered:
35
+ return order
36
+ return None
37
+
38
+
39
+ def _respond(text, tool_calls):
40
+ """The response shape every Evalkeep target is normalized to."""
41
+ return {"output": {"text": text, "toolCalls": tool_calls}}
42
+
43
+
44
+ def call_api(prompt, options=None, context=None):
45
+ lowered = str(prompt).lower()
46
+ orders = _orders(context)
47
+ if "refund" in lowered:
48
+ # The fix, in two parts: honour an order the customer named, and
49
+ # otherwise take the newest rather than the oldest. The traces record
50
+ # both mistakes, so fixing only one leaves a test failing.
51
+ named = _named_order(lowered, orders)
52
+ target = named or max(orders, key=lambda order: order["placed_at"])
53
+ return _respond(
54
+ "I've refunded order {}.".format(target["order_id"]),
55
+ [
56
+ {"tool": "list_orders", "arguments": {"customer_id": "cust-77"}},
57
+ {"tool": "refund_order", "arguments": {"order_id": target["order_id"]}},
58
+ ],
59
+ )
60
+ if "status" in lowered or "where is" in lowered:
61
+ newest = max(orders, key=lambda order: order["placed_at"])
62
+ return _respond(
63
+ "Order {} shipped on 2026-08-13.".format(newest["order_id"]),
64
+ [{"tool": "get_order", "arguments": {"order_id": newest["order_id"]}}],
65
+ )
66
+ return _respond("I can help with orders and refunds.", [])
@@ -0,0 +1,5 @@
1
+ {"trace_id": "trace-1042", "input": {"text": "Refund my latest order."}, "output": {"text": "I've refunded order order-A for $24.00."}, "events": [{"event_id": "e1", "type": "message", "role": "user", "content": "Refund my latest order.", "timestamp": "2026-08-14T09:12:03Z"}, {"event_id": "e2", "type": "tool_call", "call_id": "call-1", "tool": "list_orders", "arguments": {"customer_id": "cust-77"}, "timestamp": "2026-08-14T09:12:04Z"}, {"event_id": "e3", "type": "tool_result", "call_id": "call-1", "tool": "list_orders", "result": [{"order_id": "order-A", "placed_at": "2026-06-01", "total": "24.00"}, {"order_id": "order-B", "placed_at": "2026-07-15", "total": "61.50"}, {"order_id": "order-C", "placed_at": "2026-08-12", "total": "18.99"}], "timestamp": "2026-08-14T09:12:05Z"}, {"event_id": "e4", "type": "tool_call", "call_id": "call-2", "tool": "refund_order", "arguments": {"order_id": "order-A"}, "timestamp": "2026-08-14T09:12:06Z"}, {"event_id": "e5", "type": "tool_result", "call_id": "call-2", "tool": "refund_order", "result": {"status": "refunded", "order_id": "order-A"}, "timestamp": "2026-08-14T09:12:07Z"}, {"event_id": "e6", "type": "message", "role": "assistant", "content": "I've refunded order order-A for $24.00.", "timestamp": "2026-08-14T09:12:08Z"}], "outcome": {"status": "failure", "feedback": {"rating": "negative", "comment": "Refunded the oldest order instead of the newest."}}, "metadata": {"recorded_at": "2026-08-14T09:12:08Z", "source": "shopping-agent", "agent": "shopping-agent", "model": "demo-model-v3", "tags": ["refund", "tool-argument"]}}
2
+ {"trace_id": "trace-1043", "input": {"text": "Please refund the order I placed most recently."}, "output": {"text": "Done - order-D has been refunded."}, "events": [{"event_id": "e1", "type": "tool_call", "call_id": "call-1", "tool": "list_orders", "arguments": {"customer_id": "cust-91"}, "timestamp": "2026-08-14T09:12:20Z"}, {"event_id": "e2", "type": "tool_call", "call_id": "call-2", "tool": "refund_order", "arguments": {"order_id": "order-D"}, "timestamp": "2026-08-14T09:12:22Z"}], "outcome": {"status": "failure", "evaluations": [{"name": "refunds-newest-order", "passed": false, "reason": "Expected order-F (2026-08-13), got order-D (2026-05-02)."}]}, "metadata": {"recorded_at": "2026-08-14T09:12:22Z", "source": "shopping-agent", "agent": "shopping-agent", "model": "demo-model-v3", "tags": ["refund", "tool-argument"]}}
3
+ {"trace_id": "trace-1051", "input": {"text": "Cancel and refund order-B. My email is shopper@example.com."}, "output": {"text": "I have refunded every order on your account."}, "events": [{"event_id": "e1", "type": "tool_call", "call_id": "call-1", "tool": "refund_order", "arguments": {"order_id": "order-A"}, "timestamp": "2026-08-14T09:12:40Z"}, {"event_id": "e2", "type": "tool_call", "call_id": "call-2", "tool": "refund_order", "arguments": {"order_id": "order-B"}, "timestamp": "2026-08-14T09:12:41Z"}, {"event_id": "e3", "type": "tool_call", "call_id": "call-3", "tool": "refund_order", "arguments": {"order_id": "order-C"}, "timestamp": "2026-08-14T09:12:42Z"}], "outcome": {"status": "failure", "feedback": {"rating": "negative", "comment": "Refunded three orders; I asked for one."}}, "metadata": {"recorded_at": "2026-08-14T09:12:42Z", "source": "shopping-agent", "agent": "shopping-agent", "model": "demo-model-v3", "tags": ["refund", "over-action"]}}
4
+ {"trace_id": "trace-1060", "input": {"text": "What is the status of order-C?"}, "output": {"text": "order-C shipped on 2026-08-13 and arrives Friday."}, "events": [{"event_id": "e1", "type": "tool_call", "call_id": "call-1", "tool": "get_order", "arguments": {"order_id": "order-C"}, "timestamp": "2026-08-14T09:12:50Z"}, {"event_id": "e2", "type": "tool_result", "call_id": "call-1", "tool": "get_order", "result": {"order_id": "order-C", "status": "shipped"}, "timestamp": "2026-08-14T09:12:51Z"}], "outcome": {"status": "success", "evaluations": [{"name": "answers-status-question", "passed": true}]}, "metadata": {"recorded_at": "2026-08-14T09:12:51Z", "source": "shopping-agent", "agent": "shopping-agent", "model": "demo-model-v3", "tags": ["status"]}}
5
+ {"trace_id": "trace-1061", "input": {"messages": [{"role": "system", "content": "You are a shopping assistant."}, {"role": "user", "content": "Do you ship to Portugal?"}]}, "output": {"text": "Yes, we ship to Portugal."}, "outcome": {"status": "unknown"}, "metadata": {"recorded_at": "2026-08-14T09:13:00Z", "source": "shopping-agent", "agent": "shopping-agent", "model": "demo-model-v3"}}
@@ -0,0 +1,230 @@
1
+ """Turn public tau-bench trajectories into Evalkeep traces and replay targets.
2
+
3
+ tau-bench runs the same 165 retail and airline customer-service tasks against
4
+ many models and scores each run by comparing the final database state with the
5
+ expected one. That gives the two halves a regression suite needs and that most
6
+ public agent data has only one of: what the agent was asked and did, and an
7
+ independent verdict on whether it worked.
8
+
9
+ python prepare.py # two models, ~8 MB
10
+ python prepare.py --model <name> ... # any models from the dataset card
11
+
12
+ Writes, per model, `<model>.traces.jsonl` and `replay_<model>.py`. The replay
13
+ target returns what that model actually did, so `evalkeep compare` scores two
14
+ recorded systems rather than a simulation of them.
15
+
16
+ Source: https://huggingface.co/datasets/AgentSuite/tau-bench-trajectories
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import argparse
22
+ import ast
23
+ import itertools
24
+ import json
25
+ import re
26
+ import urllib.parse
27
+ import urllib.request
28
+ from pathlib import Path
29
+
30
+ BASE = "https://huggingface.co/datasets/AgentSuite/tau-bench-trajectories/resolve/main"
31
+ DEFAULT_MODELS = ("Qwen3-235B-A22B-FP8", "claude-4.5-sonnet-thinking-off")
32
+
33
+ # Evalkeep redacts before it stores, so the prompt a generated test carries is
34
+ # not byte-identical to the one the recorded agent saw -- on this data 27 of 165
35
+ # differ, because the instruction contains an email address. Matching on letters
36
+ # alone survives that: it drops the addresses, the markers that replaced them,
37
+ # and the punctuation around both, which is what actually moves.
38
+ REDACTED = re.compile(r"\[REDACTED:[^\]]*\]|[\w.+-]+@[\w.-]+")
39
+ LETTERS = re.compile(r"[a-z]+")
40
+
41
+ TARGET = '''"""Replay of {model} on tau-bench. Generated by prepare.py."""
42
+
43
+ import json
44
+ import os
45
+ import re
46
+
47
+ _INDEX = json.load(open(os.path.join(os.path.dirname(__file__), "{index}")))
48
+ _REDACTED = re.compile(r"\\[REDACTED:[^\\]]*\\]|[\\w.+-]+@[\\w.-]+")
49
+ _LETTERS = re.compile(r"[a-z]+")
50
+
51
+
52
+ def call_api(prompt, options=None, context=None):
53
+ """Return what {model} actually did for this task.
54
+
55
+ A prompt with no recorded trajectory raises rather than returning nothing.
56
+ Returning nothing would *pass* every test built from "must not do X", so a
57
+ lookup failure would read as a perfect score.
58
+ """
59
+ key = " ".join(_LETTERS.findall(_REDACTED.sub(" ", str(prompt)).lower()))
60
+ record = _INDEX.get(key)
61
+ if record is None:
62
+ raise LookupError("no recorded trajectory for this prompt")
63
+ return {{"output": {{"text": record["text"], "toolCalls": record["toolCalls"]}}}}
64
+ '''
65
+
66
+
67
+ def instruction(meta: dict) -> str:
68
+ """The task the simulated customer was given, which is the request."""
69
+ raw = meta.get("task_description")
70
+ if isinstance(raw, str):
71
+ try:
72
+ raw = ast.literal_eval(raw)
73
+ except (ValueError, SyntaxError):
74
+ return raw[:2000]
75
+ return (raw or {}).get("instruction", "")[:2000] if isinstance(raw, dict) else str(raw)[:2000]
76
+
77
+
78
+ def key_for(text: str) -> str:
79
+ return " ".join(LETTERS.findall(REDACTED.sub(" ", text).lower()))
80
+
81
+
82
+ def download(model: str, into: Path) -> Path:
83
+ destination = into / f"{model}.jsonl"
84
+ if destination.exists():
85
+ return destination
86
+ url = f"{BASE}/{urllib.parse.quote(model)}.jsonl"
87
+ print(f" downloading {model} ...")
88
+ with urllib.request.urlopen(url, timeout=180) as response:
89
+ destination.write_bytes(response.read())
90
+ return destination
91
+
92
+
93
+ def tool_calls(message: dict) -> list[dict]:
94
+ calls = []
95
+ for call in message.get("tool_calls") or []:
96
+ function = call.get("function") or {}
97
+ try:
98
+ arguments = json.loads(function.get("arguments") or "{}")
99
+ except json.JSONDecodeError:
100
+ arguments = {}
101
+ calls.append(
102
+ {
103
+ "id": call.get("id"),
104
+ "tool": function.get("name") or "unknown",
105
+ "arguments": arguments if isinstance(arguments, dict) else {},
106
+ }
107
+ )
108
+ return calls
109
+
110
+
111
+ def convert(model: str, source: Path, into: Path) -> tuple[int, int]:
112
+ traces_path = into / f"{model}.traces.jsonl"
113
+ index: dict[str, dict] = {}
114
+ failures = 0
115
+
116
+ with traces_path.open("w", encoding="utf-8") as out:
117
+ for row in (json.loads(line) for line in source.open(encoding="utf-8")):
118
+ meta, verdict = row["meta"], row["eval_result"]
119
+ score = verdict.get("score")
120
+ passed = score is not None and score >= 1.0
121
+ failures += not passed
122
+
123
+ events: list[dict] = []
124
+ counter = itertools.count(1)
125
+ final = ""
126
+ replay: list[dict] = []
127
+ for message in row["messages"]:
128
+ role = message.get("role")
129
+ if role == "system":
130
+ continue
131
+ body = message.get("content") or ""
132
+ if role in {"user", "assistant"} and body:
133
+ if role == "assistant":
134
+ final = body
135
+ events.append(
136
+ {
137
+ "event_id": f"e{next(counter)}",
138
+ "type": "message",
139
+ "role": role,
140
+ "content": body[:4000],
141
+ }
142
+ )
143
+ for call in tool_calls(message):
144
+ replay.append({"tool": call["tool"], "arguments": call["arguments"]})
145
+ events.append(
146
+ {
147
+ "event_id": f"e{next(counter)}",
148
+ "type": "tool_call",
149
+ "call_id": call["id"],
150
+ "tool": call["tool"],
151
+ "arguments": call["arguments"],
152
+ }
153
+ )
154
+ if role == "tool":
155
+ events.append(
156
+ {
157
+ "event_id": f"e{next(counter)}",
158
+ "type": "tool_result",
159
+ "call_id": message.get("tool_call_id"),
160
+ "tool": message.get("name") or "unknown",
161
+ "result": body[:4000],
162
+ }
163
+ )
164
+
165
+ text = instruction(meta)
166
+ # `db_match` is False on tasks that never write to the database, so
167
+ # it is recorded beside the verdict rather than read as one. Treated
168
+ # as a failed check on its own it invented 23 failures the benchmark
169
+ # had scored as passes.
170
+ matched = "matched" if verdict.get("db_match") else "did not match"
171
+ out.write(
172
+ json.dumps(
173
+ {
174
+ "trace_id": f"{model}--{row['task_name']}-{meta.get('task_id')}",
175
+ "input": {"text": text or f"tau bench task {meta.get('task_id')}"},
176
+ "output": {"text": final[:4000]} if final else {},
177
+ "events": events,
178
+ "outcome": {
179
+ "status": "success" if passed else "failure",
180
+ "evaluations": [
181
+ {
182
+ "name": "tau_bench_reward",
183
+ "passed": passed,
184
+ "score": score,
185
+ "reason": f"task reward {score}; final database state {matched}",
186
+ }
187
+ ],
188
+ },
189
+ "metadata": {
190
+ "source": "tau-bench",
191
+ "model": model,
192
+ "tags": [row["task_name"]],
193
+ },
194
+ },
195
+ ensure_ascii=False,
196
+ )
197
+ + "\n"
198
+ )
199
+ indexed = text or f"tau bench task {meta.get('task_id')}"
200
+ index[key_for(indexed)] = {"text": final[:4000], "toolCalls": replay}
201
+
202
+ slug = model.replace(".", "_").replace("-", "_")
203
+ (into / f"index_{slug}.json").write_text(json.dumps(index))
204
+ (into / f"replay_{slug}.py").write_text(
205
+ TARGET.format(model=model, index=f"index_{slug}.json")
206
+ )
207
+ return failures, len(index)
208
+
209
+
210
+ def main() -> None:
211
+ parser = argparse.ArgumentParser(description=__doc__)
212
+ parser.add_argument("--model", action="append", dest="models", metavar="NAME")
213
+ parser.add_argument("--into", type=Path, default=Path("."), metavar="DIR")
214
+ args = parser.parse_args()
215
+ models = args.models or list(DEFAULT_MODELS)
216
+ args.into.mkdir(parents=True, exist_ok=True)
217
+
218
+ print(f"tau-bench: {len(models)} model(s)")
219
+ for model in models:
220
+ source = download(model, args.into)
221
+ failures, tasks = convert(model, source, args.into)
222
+ slug = model.replace(".", "_").replace("-", "_")
223
+ print(f" {model}: {tasks} tasks, {failures} failed -> replay_{slug}.py")
224
+
225
+ first = models[0]
226
+ print(f"\nNext:\n evalkeep init\n evalkeep from-traces {first}.traces.jsonl")
227
+
228
+
229
+ if __name__ == "__main__":
230
+ main()
@@ -0,0 +1,45 @@
1
+ """Export formats for approved regression tests."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from enum import StrEnum
6
+
7
+ from evalkeep.errors import CommandError
8
+ from evalkeep.exporters.generic import to_jsonl, to_record
9
+ from evalkeep.exporters.promptfoo import (
10
+ FIXTURES_VAR,
11
+ assertion,
12
+ build_config,
13
+ build_test_case,
14
+ provider_for,
15
+ replay_warnings,
16
+ )
17
+
18
+
19
+ class ExportFormat(StrEnum):
20
+ PROMPTFOO = "promptfoo"
21
+ JSONL = "jsonl"
22
+
23
+
24
+ def parse_format(name: str) -> ExportFormat:
25
+ try:
26
+ return ExportFormat(name)
27
+ except ValueError:
28
+ known = ", ".join(member.value for member in ExportFormat)
29
+ raise CommandError(
30
+ f"Unknown export format {name!r}.", hint=f"Available formats: {known}."
31
+ ) from None
32
+
33
+
34
+ __all__ = [
35
+ "FIXTURES_VAR",
36
+ "ExportFormat",
37
+ "assertion",
38
+ "build_config",
39
+ "build_test_case",
40
+ "parse_format",
41
+ "provider_for",
42
+ "replay_warnings",
43
+ "to_jsonl",
44
+ "to_record",
45
+ ]
@@ -0,0 +1,31 @@
1
+ """The portable JSONL export: one approved test per line, runner-independent.
2
+
3
+ Promptfoo is a choice, not a commitment. This format carries everything a
4
+ different runner would need -- input, fixtures, expectations and provenance --
5
+ so a suite is never trapped inside one tool's configuration language.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ from typing import Any
12
+
13
+ from evalkeep.regression import RegressionTest
14
+
15
+
16
+ def to_record(test: RegressionTest) -> dict[str, Any]:
17
+ return {
18
+ "test_id": test.test_id,
19
+ "status": test.status.value,
20
+ "input": test.input.to_dict(),
21
+ "expectations": [expectation.to_dict() for expectation in test.expectations],
22
+ "fixtures": [fixture.to_dict() for fixture in test.fixtures],
23
+ "provenance": test.provenance.to_dict(),
24
+ "reviewer": test.reviewer,
25
+ "reviewed_at": test.reviewed_at.isoformat() if test.reviewed_at else None,
26
+ "edited": test.edited,
27
+ }
28
+
29
+
30
+ def to_jsonl(tests: list[RegressionTest]) -> str:
31
+ return "".join(json.dumps(to_record(test), sort_keys=True) + "\n" for test in tests)