eval-builder 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
eval_builder/draft.py ADDED
@@ -0,0 +1,271 @@
1
+ """Write case and rubric skeletons for the agent to fill in, and validate them.
2
+
3
+ The tool never writes expected behavior itself. It copies the observed input and
4
+ output from the selected traces and leaves TODO markers that validation refuses
5
+ to accept on cases marked `ready`.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from pathlib import Path
11
+ from typing import Any
12
+
13
+ from .ingest import load_traces
14
+ from .io import read_json, read_yaml, write_yaml
15
+ from .workspace import Workspace
16
+
17
+ STATUSES = ("draft", "ready", "dropped")
18
+ MODES = ("pointwise", "pairwise")
19
+ TODO = "TODO"
20
+
21
+ CASES_HEADER = """eval-builder case file. Fill in expected_behavior and criteria with the user,
22
+ then set status: ready (or dropped). Validation rejects ready cases that still
23
+ contain TODO. Criteria ids must exist in rubric.yaml.
24
+ Optional per case: reference_output (an ideal answer) and compare_output (a second
25
+ output for pairwise judging, for example from a new prompt version)."""
26
+
27
+ RUBRIC_HEADER = """eval-builder rubric. Define the criteria your cases reference and the judges
28
+ you plan to check. A judge prompt may use {input}, {context} (earlier turns), {output},
29
+ {answer_a}, {answer_b}, {expected_behavior} and {criteria}; judge-plan renders it per
30
+ request."""
31
+
32
+ UPDATABLE = (
33
+ "expected_behavior",
34
+ "criteria",
35
+ "reference_output",
36
+ "compare_output",
37
+ "status",
38
+ "notes",
39
+ "tags",
40
+ )
41
+
42
+
43
+ def _rubric_skeleton() -> dict[str, Any]:
44
+ return {
45
+ "version": 1,
46
+ "criteria": [
47
+ {
48
+ "id": "TODO-criterion-id",
49
+ "description": "TODO: what must be true of a good answer",
50
+ "scale": "pass_fail",
51
+ },
52
+ ],
53
+ "judges": [
54
+ {
55
+ "id": "TODO-judge-id",
56
+ "mode": "pointwise",
57
+ "criteria": ["TODO-criterion-id"],
58
+ "labels": ["pass", "fail"],
59
+ "prompt": "TODO: the exact judge prompt. Use {input}, {output}, "
60
+ "{expected_behavior}, {criteria}.",
61
+ },
62
+ ],
63
+ }
64
+
65
+
66
+ def draft(
67
+ workspace: str | Path, suite: str = "eval-builder suite", force: bool = False
68
+ ) -> dict[str, Any]:
69
+ ws = Workspace.at(workspace)
70
+ if not ws.selection.exists():
71
+ raise FileNotFoundError(f"{ws.selection} not found; run `eval-builder select` first")
72
+ selection = read_json(ws.selection)
73
+ traces = {t.id: t for t in load_traces(ws.root)}
74
+ existing: dict[str, Any] = {}
75
+ if ws.cases.exists() and not force:
76
+ existing = read_yaml(ws.cases) or {}
77
+ cases = list(existing.get("cases") or [])
78
+ known = {c.get("trace_id") for c in cases}
79
+ next_num = len(cases) + 1
80
+ added = 0
81
+ for pick in selection["selected"]:
82
+ tid = pick["trace_id"]
83
+ if tid in known:
84
+ continue
85
+ t = traces.get(tid)
86
+ if t is None:
87
+ continue
88
+ case: dict[str, Any] = {
89
+ "id": f"case-{next_num:03d}",
90
+ "trace_id": tid,
91
+ "status": "draft",
92
+ "selected_because": pick["reasons"],
93
+ "input": t.input,
94
+ }
95
+ ctx = t.prior_turns()
96
+ if ctx:
97
+ case["context"] = ctx
98
+ case["observed_output"] = t.output
99
+ case["observed_failure"] = bool(pick.get("failure"))
100
+ case["expected_behavior"] = f"{TODO}: one or two sentences on what a good answer does"
101
+ case["criteria"] = ["TODO-criterion-id"]
102
+ case["reference_output"] = None
103
+ tags = []
104
+ if t.route:
105
+ tags.append(f"route:{t.route}")
106
+ if t.model:
107
+ tags.append(f"model:{t.model}")
108
+ tags += [f"tool:{x}" for x in t.tools]
109
+ if t.error:
110
+ tags.append("error")
111
+ if t.feedback:
112
+ tags.append(f"feedback:{t.feedback}")
113
+ tags.append(f"cluster:{pick['cluster']}")
114
+ case["tags"] = tags
115
+ cases.append(case)
116
+ known.add(tid)
117
+ next_num += 1
118
+ added += 1
119
+ doc = {"suite": existing.get("suite", suite), "version": 1, "cases": cases}
120
+ write_yaml(ws.cases, doc, CASES_HEADER)
121
+ rubric_created = False
122
+ if not ws.rubric.exists() or force:
123
+ write_yaml(ws.rubric, _rubric_skeleton(), RUBRIC_HEADER)
124
+ rubric_created = True
125
+ return {
126
+ "cases_file": str(ws.cases),
127
+ "rubric_file": str(ws.rubric),
128
+ "cases_total": len(cases),
129
+ "cases_added": added,
130
+ "rubric_created": rubric_created,
131
+ "next": "fill expected_behavior and criteria per case, define criteria and judges "
132
+ "in rubric.yaml, then run `eval-builder validate`",
133
+ }
134
+
135
+
136
+ def load_cases(workspace: str | Path) -> dict[str, Any]:
137
+ ws = Workspace.at(workspace)
138
+ if not ws.cases.exists():
139
+ raise FileNotFoundError(f"{ws.cases} not found; run `eval-builder draft` first")
140
+ return read_yaml(ws.cases) or {"cases": []}
141
+
142
+
143
+ def load_rubric(workspace: str | Path) -> dict[str, Any]:
144
+ ws = Workspace.at(workspace)
145
+ if not ws.rubric.exists():
146
+ raise FileNotFoundError(f"{ws.rubric} not found; run `eval-builder draft` first")
147
+ return read_yaml(ws.rubric) or {}
148
+
149
+
150
+ def _has_todo(v: Any) -> bool:
151
+ if isinstance(v, str):
152
+ return TODO in v
153
+ if isinstance(v, list):
154
+ return any(_has_todo(x) for x in v)
155
+ if isinstance(v, dict):
156
+ return any(_has_todo(x) for x in v.values())
157
+ return False
158
+
159
+
160
+ def validate_rubric(rubric: dict[str, Any]) -> list[dict[str, str]]:
161
+ errors: list[dict[str, str]] = []
162
+
163
+ def err(where: str, problem: str) -> None:
164
+ errors.append({"where": where, "problem": problem})
165
+
166
+ crit = rubric.get("criteria") or []
167
+ if not crit:
168
+ err("rubric.criteria", "no criteria defined")
169
+ seen: set[str] = set()
170
+ for i, c in enumerate(crit):
171
+ cid = str(c.get("id", ""))
172
+ if not cid or _has_todo(cid):
173
+ err(f"rubric.criteria[{i}].id", "missing or still TODO")
174
+ if cid in seen:
175
+ err(f"rubric.criteria[{i}].id", f"duplicate id {cid}")
176
+ seen.add(cid)
177
+ if not str(c.get("description", "")).strip() or _has_todo(c.get("description")):
178
+ err(f"rubric.criteria[{i}].description", "missing or still TODO")
179
+ judge_ids: set[str] = set()
180
+ for i, j in enumerate(rubric.get("judges") or []):
181
+ jid = str(j.get("id", ""))
182
+ if not jid or _has_todo(jid):
183
+ err(f"rubric.judges[{i}].id", "missing or still TODO")
184
+ if jid in judge_ids:
185
+ err(f"rubric.judges[{i}].id", f"duplicate id {jid}")
186
+ judge_ids.add(jid)
187
+ if j.get("mode", "pointwise") not in MODES:
188
+ err(f"rubric.judges[{i}].mode", f"must be one of {', '.join(MODES)}")
189
+ if not j.get("labels"):
190
+ err(f"rubric.judges[{i}].labels", "list the verdict labels the judge may return")
191
+ if _has_todo(j.get("prompt")):
192
+ err(
193
+ f"rubric.judges[{i}].prompt",
194
+ "still TODO (remove the key if you run judges with your own prompt elsewhere)",
195
+ )
196
+ for cid in j.get("criteria") or []:
197
+ if cid not in seen:
198
+ err(f"rubric.judges[{i}].criteria", f"unknown criterion {cid}")
199
+ return errors
200
+
201
+
202
+ def validate(workspace: str | Path) -> dict[str, Any]:
203
+ doc = load_cases(workspace)
204
+ rubric = load_rubric(workspace)
205
+ errors = validate_rubric(rubric)
206
+ crit_ids = {str(c.get("id")) for c in rubric.get("criteria") or []}
207
+ counts = dict.fromkeys(STATUSES, 0)
208
+ ids: set[str] = set()
209
+ for i, c in enumerate(doc.get("cases") or []):
210
+ cid = str(c.get("id", f"#{i}"))
211
+ where = f"cases[{cid}]"
212
+ if cid in ids:
213
+ errors.append({"where": where, "problem": "duplicate case id"})
214
+ ids.add(cid)
215
+ status = c.get("status", "draft")
216
+ if status not in STATUSES:
217
+ errors.append({"where": where, "problem": f"status must be one of {STATUSES}"})
218
+ continue
219
+ counts[status] += 1
220
+ if status != "ready":
221
+ continue
222
+ if not str(c.get("input", "")).strip():
223
+ errors.append({"where": where, "problem": "empty input"})
224
+ eb = c.get("expected_behavior")
225
+ if not isinstance(eb, str) or not eb.strip() or _has_todo(eb):
226
+ errors.append(
227
+ {"where": f"{where}.expected_behavior", "problem": "missing or still TODO"}
228
+ )
229
+ crit = c.get("criteria") or []
230
+ if not crit:
231
+ errors.append({"where": f"{where}.criteria", "problem": "no criteria"})
232
+ for k in crit:
233
+ if k not in crit_ids:
234
+ errors.append({"where": f"{where}.criteria", "problem": f"unknown criterion {k}"})
235
+ if _has_todo(c.get("reference_output")):
236
+ errors.append({"where": f"{where}.reference_output", "problem": "still TODO"})
237
+ if counts["ready"] == 0:
238
+ errors.append({"where": "cases", "problem": "no cases have status: ready"})
239
+ warnings = []
240
+ if counts["draft"]:
241
+ warnings.append(f"{counts['draft']} case(s) still draft; they are not exported")
242
+ return {"valid": not errors, "counts": counts, "errors": errors, "warnings": warnings}
243
+
244
+
245
+ def update_case(workspace: str | Path, case_id: str, **fields: Any) -> dict[str, Any]:
246
+ ws = Workspace.at(workspace)
247
+ doc = load_cases(ws.root)
248
+ bad = [k for k in fields if k not in UPDATABLE]
249
+ if bad:
250
+ raise ValueError(f"cannot update {bad}; allowed: {', '.join(UPDATABLE)}")
251
+ if "status" in fields and fields["status"] not in STATUSES:
252
+ raise ValueError(f"status must be one of {STATUSES}")
253
+ for c in doc.get("cases") or []:
254
+ if c.get("id") == case_id:
255
+ for k, v in fields.items():
256
+ if v is not None:
257
+ c[k] = v
258
+ write_yaml(ws.cases, doc, CASES_HEADER)
259
+ return c
260
+ raise KeyError(f"no case with id {case_id}")
261
+
262
+
263
+ def write_rubric(workspace: str | Path, rubric: dict[str, Any]) -> dict[str, Any]:
264
+ ws = Workspace.at(workspace)
265
+ rubric = {"version": 1, **rubric}
266
+ write_yaml(ws.rubric, rubric, RUBRIC_HEADER)
267
+ return {"rubric_file": str(ws.rubric), "errors": validate_rubric(rubric)}
268
+
269
+
270
+ def ready_cases(workspace: str | Path) -> list[dict[str, Any]]:
271
+ return [c for c in load_cases(workspace).get("cases") or [] if c.get("status") == "ready"]
eval_builder/export.py ADDED
@@ -0,0 +1,284 @@
1
+ """Export ready cases to the eval tools people already use.
2
+
3
+ eval-builder is not an eval runner. It writes files for promptfoo, DeepEval and
4
+ Inspect AI (and plain JSONL) so the suite runs where your team already works.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ from pathlib import Path
11
+ from typing import Any
12
+
13
+ from .draft import load_cases, load_rubric, validate
14
+ from .io import read_json, sha256_file, write_json, write_jsonl, write_yaml
15
+ from .workspace import Workspace
16
+
17
+ FORMATS = ("promptfoo", "deepeval", "inspect", "jsonl")
18
+
19
+
20
+ def _criteria_text(case: dict[str, Any], crit: dict[str, str]) -> str:
21
+ return "\n".join(f"- {c}: {crit.get(c, '')}" for c in case.get("criteria") or [])
22
+
23
+
24
+ def _metadata(case: dict[str, Any]) -> dict[str, Any]:
25
+ return {
26
+ "case_id": case["id"],
27
+ "trace_id": case.get("trace_id"),
28
+ "tags": case.get("tags") or [],
29
+ "criteria": case.get("criteria") or [],
30
+ }
31
+
32
+
33
+ def _judge_summary(ws: Workspace) -> dict[str, Any] | None:
34
+ if not ws.judge_check.exists():
35
+ return None
36
+ jc = read_json(ws.judge_check)
37
+ return {
38
+ "trustworthy": jc.get("trustworthy", []),
39
+ "verdicts": {s["judge"]: s["verdict"] for s in jc.get("summary", [])},
40
+ "thresholds": jc.get("thresholds"),
41
+ }
42
+
43
+
44
+ def _export_jsonl(cases: list[dict[str, Any]], crit: dict[str, str], out: Path) -> list[Path]:
45
+ rows = []
46
+ for c in cases:
47
+ rows.append(
48
+ {
49
+ "id": c["id"],
50
+ "input": c.get("input"),
51
+ "context": c.get("context") or [],
52
+ "expected_behavior": c.get("expected_behavior"),
53
+ "criteria": [
54
+ {"id": k, "description": crit.get(k, "")} for k in c.get("criteria") or []
55
+ ],
56
+ "reference_output": c.get("reference_output"),
57
+ "observed_output": c.get("observed_output"),
58
+ "metadata": _metadata(c),
59
+ }
60
+ )
61
+ path = out / "cases.jsonl"
62
+ write_jsonl(path, rows)
63
+ return [path]
64
+
65
+
66
+ PROMPTFOO_HEADER = """Generated by eval-builder. Replace the `echo` provider with your app's
67
+ provider (https://www.promptfoo.dev/docs/providers/). `echo` only returns the prompt, so a run
68
+ with it checks the plumbing, not your app. llm-rubric assertions need a grader model
69
+ configured in promptfoo (defaultTest.options.provider)."""
70
+
71
+
72
+ def _export_promptfoo(
73
+ cases: list[dict[str, Any]], crit: dict[str, str], suite: str, out: Path
74
+ ) -> list[Path]:
75
+ tests = []
76
+ for c in cases:
77
+ variables: dict[str, Any] = {"input": c.get("input")}
78
+ if c.get("context"):
79
+ variables["context"] = json.dumps(c["context"], ensure_ascii=False)
80
+ rubric = f"{c.get('expected_behavior', '').strip()}\n\nCriteria:\n{_criteria_text(c, crit)}"
81
+ test: dict[str, Any] = {
82
+ "description": f"{c['id']}: {str(c.get('input', ''))[:60].strip()}",
83
+ "vars": variables,
84
+ "assert": [{"type": "llm-rubric", "value": rubric}],
85
+ "metadata": {
86
+ "case_id": c["id"],
87
+ "trace_id": c.get("trace_id"),
88
+ "criteria": ",".join(c.get("criteria") or []),
89
+ },
90
+ }
91
+ tests.append(test)
92
+ config = {"description": suite, "prompts": ["{{input}}"], "providers": ["echo"], "tests": tests}
93
+ path = out / "promptfooconfig.yaml"
94
+ write_yaml(path, config, PROMPTFOO_HEADER)
95
+ return [path]
96
+
97
+
98
+ DEEPEVAL_TEST = '''"""Generated by eval-builder. Run with: deepeval test run test_eval_builder.py
99
+
100
+ Fill in `your_app` to call the system under test. Set EVAL_BUILDER_USE_OBSERVED=1 to
101
+ score the outputs that were logged instead (a regression check on past behavior).
102
+ GEval needs a judge model configured in DeepEval.
103
+ """
104
+
105
+ import json
106
+ import os
107
+ from pathlib import Path
108
+
109
+ import pytest
110
+ from deepeval import assert_test
111
+ from deepeval.metrics import GEval
112
+ from deepeval.test_case import LLMTestCase
113
+
114
+ try: # DeepEval 4.x
115
+ from deepeval.test_case import SingleTurnParams as Params
116
+ except ImportError: # older DeepEval
117
+ from deepeval.test_case import LLMTestCaseParams as Params
118
+
119
+ CASES = json.loads((Path(__file__).parent / "dataset.json").read_text(encoding="utf-8"))
120
+
121
+
122
+ def your_app(input: str) -> str:
123
+ raise NotImplementedError("call the app under test here")
124
+
125
+
126
+ def make_metric(criteria: str) -> GEval:
127
+ return GEval(
128
+ name="expected behavior",
129
+ criteria=(
130
+ "Judge whether the actual output does what the expected output describes. "
131
+ "The expected output is a description of good behavior, not an exact answer. "
132
+ + criteria
133
+ ),
134
+ evaluation_params=[
135
+ Params.INPUT,
136
+ Params.ACTUAL_OUTPUT,
137
+ Params.EXPECTED_OUTPUT,
138
+ ],
139
+ )
140
+
141
+
142
+ @pytest.mark.parametrize("case", CASES, ids=[c["name"] for c in CASES])
143
+ def test_case(case):
144
+ if os.environ.get("EVAL_BUILDER_USE_OBSERVED") == "1":
145
+ actual = case["actual_output"]
146
+ else:
147
+ actual = your_app(case["input"])
148
+ test_case = LLMTestCase(
149
+ input=case["input"],
150
+ actual_output=actual,
151
+ expected_output=case["expected_output"],
152
+ context=case.get("context") or None,
153
+ )
154
+ assert_test(test_case, [make_metric(case["additional_metadata"]["criteria_text"])])
155
+ '''
156
+
157
+
158
+ def _export_deepeval(cases: list[dict[str, Any]], crit: dict[str, str], out: Path) -> list[Path]:
159
+ rows = []
160
+ for c in cases:
161
+ ctx = [f"{m['role']}: {m['content']}" for m in c.get("context") or []]
162
+ rows.append(
163
+ {
164
+ "name": c["id"],
165
+ "input": c.get("input"),
166
+ "expected_output": c.get("expected_behavior"),
167
+ "actual_output": c.get("observed_output"),
168
+ "context": ctx or None,
169
+ "additional_metadata": {**_metadata(c), "criteria_text": _criteria_text(c, crit)},
170
+ }
171
+ )
172
+ data = out / "dataset.json"
173
+ write_json(data, rows)
174
+ test = out / "test_eval_builder.py"
175
+ test.write_text(DEEPEVAL_TEST, encoding="utf-8")
176
+ return [data, test]
177
+
178
+
179
+ INSPECT_TASK = '''"""Generated by eval-builder.
180
+
181
+ Run with: inspect eval task.py --model <provider/model>
182
+
183
+ `target` holds the expected behavior written for each case; model_graded_qa asks a
184
+ grader model whether the answer meets it.
185
+ """
186
+
187
+ from pathlib import Path
188
+
189
+ from inspect_ai import Task, task
190
+ from inspect_ai.dataset import json_dataset
191
+ from inspect_ai.scorer import model_graded_qa
192
+ from inspect_ai.solver import generate
193
+
194
+
195
+ @task
196
+ def eval_builder_suite():
197
+ return Task(
198
+ dataset=json_dataset(str(Path(__file__).parent / "dataset.jsonl")),
199
+ solver=generate(),
200
+ scorer=model_graded_qa(),
201
+ )
202
+ '''
203
+
204
+
205
+ def _export_inspect(cases: list[dict[str, Any]], crit: dict[str, str], out: Path) -> list[Path]:
206
+ rows = []
207
+ for c in cases:
208
+ ctx = c.get("context") or []
209
+ inp: Any = c.get("input")
210
+ if ctx:
211
+ inp = [
212
+ {"role": m["role"], "content": m["content"]}
213
+ for m in ctx
214
+ if m.get("role") in ("system", "user", "assistant")
215
+ ]
216
+ inp.append({"role": "user", "content": c.get("input")})
217
+ target = c.get("expected_behavior", "").strip()
218
+ criteria = _criteria_text(c, crit)
219
+ if criteria:
220
+ target = f"{target}\n\nCriteria:\n{criteria}"
221
+ rows.append(
222
+ {
223
+ "id": c["id"],
224
+ "input": inp,
225
+ "target": target,
226
+ "metadata": {**_metadata(c), "observed_output": c.get("observed_output")},
227
+ }
228
+ )
229
+ data = out / "dataset.jsonl"
230
+ write_jsonl(data, rows)
231
+ task_file = out / "task.py"
232
+ task_file.write_text(INSPECT_TASK, encoding="utf-8")
233
+ return [data, task_file]
234
+
235
+
236
+ def export(workspace: str | Path, formats: list[str] | None = None) -> dict[str, Any]:
237
+ ws = Workspace.at(workspace)
238
+ formats = formats or list(FORMATS)
239
+ bad = [f for f in formats if f not in FORMATS]
240
+ if bad:
241
+ raise ValueError(f"unknown export format(s) {bad}; expected {', '.join(FORMATS)}")
242
+ check = validate(ws.root)
243
+ if not check["valid"]:
244
+ return {
245
+ "exported": False,
246
+ "validation": check,
247
+ "next": "fix the validation errors, then export again",
248
+ }
249
+ doc = load_cases(ws.root)
250
+ rubric = load_rubric(ws.root)
251
+ crit = {str(c["id"]): str(c.get("description", "")) for c in rubric.get("criteria") or []}
252
+ cases = [c for c in doc.get("cases") or [] if c.get("status") == "ready"]
253
+ suite = str(doc.get("suite", "eval-builder suite"))
254
+ files: list[Path] = []
255
+ for f in formats:
256
+ out = ws.exports / f
257
+ out.mkdir(parents=True, exist_ok=True)
258
+ if f == "jsonl":
259
+ files += _export_jsonl(cases, crit, out)
260
+ elif f == "promptfoo":
261
+ files += _export_promptfoo(cases, crit, suite, out)
262
+ elif f == "deepeval":
263
+ files += _export_deepeval(cases, crit, out)
264
+ elif f == "inspect":
265
+ files += _export_inspect(cases, crit, out)
266
+ judges = _judge_summary(ws)
267
+ trusted_prompts = {}
268
+ if judges:
269
+ for j in rubric.get("judges") or []:
270
+ if j.get("id") in judges["trustworthy"] and j.get("prompt"):
271
+ trusted_prompts[j["id"]] = {
272
+ "mode": j.get("mode"),
273
+ "labels": j.get("labels"),
274
+ "prompt": j["prompt"],
275
+ }
276
+ manifest = {
277
+ "cases": len(cases),
278
+ "formats": formats,
279
+ "files": [{"path": str(p.relative_to(ws.root)), "sha256": sha256_file(p)} for p in files],
280
+ "judge_check": judges,
281
+ "trusted_judge_prompts": trusted_prompts,
282
+ }
283
+ write_json(ws.exports / "manifest.json", manifest)
284
+ return {"exported": True, **manifest}