eval-builder 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,3 @@
1
+ """eval-builder: turn real LLM app logs into an eval suite and check which judges you can trust."""
2
+
3
+ __version__ = "0.1.0"
eval_builder/cli.py ADDED
@@ -0,0 +1,354 @@
1
+ """Command line interface. Every command accepts --json for machine-readable output."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ import sys
8
+ from typing import Any
9
+
10
+ from . import __version__
11
+
12
+
13
+ def _csv(s: str | None) -> list[str] | None:
14
+ return [x.strip() for x in s.split(",") if x.strip()] if s else None
15
+
16
+
17
+ def _print(result: Any, as_json: bool, human: str) -> None:
18
+ if as_json:
19
+ print(json.dumps(result, indent=2, ensure_ascii=False))
20
+ else:
21
+ print(human)
22
+
23
+
24
+ def _cmd_ingest(a: argparse.Namespace) -> int:
25
+ from .ingest import ingest
26
+
27
+ r = ingest(a.paths, a.workspace, a.format, redact=not a.no_redact)
28
+ lines = [f"ingested {r['traces']} traces into {r['output_file']}"]
29
+ for s in r["sources"]:
30
+ skipped = sum(s["skipped"].values())
31
+ lines.append(
32
+ f" {s['path']}: format={s['format']} records={s['records']} "
33
+ f"traces={s['traces']} skipped={skipped} sha256={s['sha256'][:16]}"
34
+ )
35
+ red = r["redactions"]
36
+ lines.append(
37
+ f"redactions: {red['total']} "
38
+ + (str(red["by_kind"]) if red["enabled"] else "(redaction off)")
39
+ )
40
+ _print(r, a.json, "\n".join(lines))
41
+ return 0
42
+
43
+
44
+ def _cmd_select(a: argparse.Namespace) -> int:
45
+ from .select import select
46
+
47
+ r = select(
48
+ a.workspace,
49
+ n=a.n,
50
+ seed=a.seed,
51
+ clusters=a.clusters,
52
+ failure_share=a.failure_share,
53
+ near_dup_threshold=a.near_dup,
54
+ dedupe_on=a.dedupe_on,
55
+ stratify=_csv(a.stratify),
56
+ )
57
+ p = r["population"]
58
+ lines = [
59
+ f"{p['traces']} traces -> {p['unique']} unique "
60
+ f"({p['exact_duplicates_removed']} exact dupes, {p['near_duplicates_merged']} near "
61
+ f"dupes) -> selected {r['selected_count']} ({r['selected_failures']} failures) "
62
+ f"across {r['params']['clusters']} clusters"
63
+ ]
64
+ for s in r["selected"][:10]:
65
+ lines.append(f" {s['trace_id']}: {s['reasons'][0]}")
66
+ if r["selected_count"] > 10:
67
+ lines.append(f" ... {r['selected_count'] - 10} more in selection.json")
68
+ _print(r, a.json, "\n".join(lines))
69
+ return 0
70
+
71
+
72
+ def _cmd_draft(a: argparse.Namespace) -> int:
73
+ from .draft import draft
74
+
75
+ r = draft(a.workspace, a.suite, force=a.force)
76
+ _print(
77
+ r,
78
+ a.json,
79
+ f"{r['cases_added']} case(s) added, {r['cases_total']} total in "
80
+ f"{r['cases_file']}\nnext: {r['next']}",
81
+ )
82
+ return 0
83
+
84
+
85
+ def _cmd_validate(a: argparse.Namespace) -> int:
86
+ from .draft import validate
87
+
88
+ r = validate(a.workspace)
89
+ lines = [("valid" if r["valid"] else "INVALID") + f": {r['counts']}"]
90
+ lines += [f" error {e['where']}: {e['problem']}" for e in r["errors"][:30]]
91
+ lines += [f" warning: {w}" for w in r["warnings"]]
92
+ _print(r, a.json, "\n".join(lines))
93
+ return 0 if r["valid"] else 1
94
+
95
+
96
+ def _cmd_judge_plan(a: argparse.Namespace) -> int:
97
+ from .judge.plan import judge_plan
98
+
99
+ r = judge_plan(a.workspace, _csv(a.judges), a.trials, _csv(a.probes), a.probe_trials)
100
+ _print(
101
+ r,
102
+ a.json,
103
+ f"{r['requests']} judge requests in {r['requests_file']} "
104
+ f"{r['per_judge']}\nnext: {r['next']}",
105
+ )
106
+ return 0
107
+
108
+
109
+ def _cmd_judge_run(a: argparse.Namespace) -> int:
110
+ from .judge.run import PluginDisabled, judge_run
111
+
112
+ commands = {}
113
+ for spec in a.command:
114
+ if "=" not in spec:
115
+ print(f"--command must look like judge_id=command, got {spec!r}", file=sys.stderr)
116
+ return 2
117
+ jid, cmd = spec.split("=", 1)
118
+ commands[jid.strip()] = cmd.strip()
119
+ try:
120
+ r = judge_run(
121
+ a.workspace,
122
+ commands,
123
+ enable=a.enable_judge_plugin,
124
+ timeout=a.timeout,
125
+ resume=not a.restart,
126
+ )
127
+ except PluginDisabled as e:
128
+ print(str(e), file=sys.stderr)
129
+ return 3
130
+ lines = [
131
+ f"{jid}: {e['ok']} ok, {e['errors']} errors, {e['seconds']} s"
132
+ for jid, e in r["judges"].items()
133
+ ]
134
+ _print(r, a.json, "\n".join(lines))
135
+ return 0
136
+
137
+
138
+ def _cmd_judge_check(a: argparse.Namespace) -> int:
139
+ from .judge.check import Thresholds, judge_check
140
+
141
+ th = Thresholds(
142
+ min_trials=a.min_trials,
143
+ min_cases=a.min_cases,
144
+ max_flip_rate=a.max_flip_rate,
145
+ min_position_consistency=a.min_position_consistency,
146
+ max_toward_padded_rate=a.max_toward_padded,
147
+ min_kappa=a.min_kappa,
148
+ min_labeled=a.min_labeled,
149
+ )
150
+ r = judge_check(a.workspace, a.judgments, a.labels, th)
151
+ lines = []
152
+ for s in r["summary"]:
153
+
154
+ def f(x: float | None) -> str:
155
+ return "n/a" if x is None else f"{x:.2f}"
156
+
157
+ lines.append(
158
+ f"{s['judge']:<28} {s['verdict']:<16} flip={f(s['flip_rate'])} "
159
+ f"kappa={f(s['kappa'])} acc={f(s['accuracy'])} "
160
+ f"pos={f(s['position_consistency'])} pad={f(s['toward_padded'])}"
161
+ )
162
+ lines += [f" {x}" for x in s["reasons"]]
163
+ out = {k: v for k, v in r.items() if k != "judges"} if not a.full else r
164
+ _print(out, a.json, "\n".join(lines))
165
+ return 0
166
+
167
+
168
+ def _cmd_export(a: argparse.Namespace) -> int:
169
+ from .export import export
170
+
171
+ r = export(a.workspace, _csv(a.formats))
172
+ if not r["exported"]:
173
+ errs = r["validation"]["errors"]
174
+ lines = ["not exported: validation failed"] + [
175
+ f" {e['where']}: {e['problem']}" for e in errs[:30]
176
+ ]
177
+ _print(r, a.json, "\n".join(lines))
178
+ return 1
179
+ lines = [f"exported {r['cases']} cases"] + [f" {f['path']}" for f in r["files"]]
180
+ _print(r, a.json, "\n".join(lines))
181
+ return 0
182
+
183
+
184
+ def _cmd_report(a: argparse.Namespace) -> int:
185
+ from .report import build_report
186
+
187
+ r = build_report(a.workspace, a.title)
188
+ _print(r, a.json, f"wrote {r['report_md']} and {r['report_json']}")
189
+ return 0
190
+
191
+
192
+ def _cmd_status(a: argparse.Namespace) -> int:
193
+ from .status import status
194
+
195
+ r = status(a.workspace)
196
+ lines = [f"{k}: {'done' if v else '-'}" for k, v in r["steps"].items()]
197
+ lines.append(f"next: {r['next']}")
198
+ _print(r, a.json, "\n".join(lines))
199
+ return 0
200
+
201
+
202
+ def _cmd_setup(a: argparse.Namespace) -> int:
203
+ from .setup_agents import setup
204
+
205
+ server = a.server_command.split() if a.server_command else None
206
+ r = setup(yes=a.yes, server=server)
207
+ lines = []
208
+ for act in r["actions"]:
209
+ lines.append(f"[{act['agent']}] {act['kind']}: {act['target']}")
210
+ lines.append(f" {act['detail'].strip()}")
211
+ if "result" in act:
212
+ lines.append(f" -> {act['result']}")
213
+ lines.append("applied" if r["applied"] else r["next"])
214
+ _print(r, a.json, "\n".join(lines))
215
+ return 0
216
+
217
+
218
+ def _cmd_mcp(a: argparse.Namespace) -> int:
219
+ from .mcp_server import main
220
+
221
+ main()
222
+ return 0
223
+
224
+
225
+ def build_parser() -> argparse.ArgumentParser:
226
+ p = argparse.ArgumentParser(
227
+ prog="eval-builder",
228
+ description="Turn real LLM app logs into an eval suite and check which LLM judges you "
229
+ "can trust. Runs locally; never calls a model provider.",
230
+ )
231
+ p.add_argument("--version", action="version", version=f"eval-builder {__version__}")
232
+ sub = p.add_subparsers(dest="command", required=True)
233
+
234
+ def cmd(name: str, help_: str) -> argparse.ArgumentParser:
235
+ sp = sub.add_parser(name, help=help_, description=help_)
236
+ sp.add_argument(
237
+ "-w", "--workspace", default="evalset", help="workspace directory (default: ./evalset)"
238
+ )
239
+ sp.add_argument("--json", action="store_true", help="print JSON")
240
+ return sp
241
+
242
+ s = cmd("ingest", "Read and normalize traces, redacting secrets and PII by default.")
243
+ s.add_argument("paths", nargs="+", help="log files or directories (.json, .jsonl)")
244
+ s.add_argument(
245
+ "-f",
246
+ "--format",
247
+ choices=["openai", "anthropic", "langfuse", "otel", "generic"],
248
+ help="force a format (default: detect per file)",
249
+ )
250
+ s.add_argument("--no-redact", action="store_true", help="keep emails, keys and numbers as-is")
251
+ s.set_defaults(func=_cmd_ingest)
252
+
253
+ s = cmd("select", "Pick a diverse, representative set of traces without an LLM.")
254
+ s.add_argument("-n", type=int, default=50, help="number of cases to pick (default 50)")
255
+ s.add_argument("--seed", type=int, default=0)
256
+ s.add_argument("--clusters", type=int, help="k for k-means (default: sqrt of unique traces)")
257
+ s.add_argument(
258
+ "--failure-share",
259
+ type=float,
260
+ default=0.3,
261
+ help="share of the budget reserved for errors and negative feedback",
262
+ )
263
+ s.add_argument(
264
+ "--near-dup",
265
+ type=float,
266
+ default=0.9,
267
+ help="cosine threshold for near-duplicates (default 0.9)",
268
+ )
269
+ s.add_argument("--dedupe-on", choices=["input", "input+output"], default="input")
270
+ s.add_argument("--stratify", help="extra metadata keys to cover, comma separated")
271
+ s.set_defaults(func=_cmd_select)
272
+
273
+ s = cmd("draft", "Write cases.yaml and rubric.yaml skeletons for the agent to fill in.")
274
+ s.add_argument("--suite", default="eval-builder suite", help="suite name")
275
+ s.add_argument("--force", action="store_true", help="overwrite existing cases and rubric")
276
+ s.set_defaults(func=_cmd_draft)
277
+
278
+ s = cmd("validate", "Check cases.yaml and rubric.yaml.")
279
+ s.set_defaults(func=_cmd_validate)
280
+
281
+ s = cmd("judge-plan", "List every judge call to make, with repeats and bias probes.")
282
+ s.add_argument("--judges", help="judge ids from rubric.yaml, comma separated (default all)")
283
+ s.add_argument("--trials", type=int, default=5, help="repeats per case (default 5)")
284
+ s.add_argument("--probes", help="comma separated: swap (pairwise only), pad")
285
+ s.add_argument("--probe-trials", type=int, default=3, help="repeats per probe (default 3)")
286
+ s.set_defaults(func=_cmd_judge_plan)
287
+
288
+ s = cmd("judge-run", "Run judge requests through YOUR command (off by default).")
289
+ s.add_argument(
290
+ "--command",
291
+ action="append",
292
+ required=True,
293
+ help="judge_id=command; the command reads JSON lines on stdin and writes "
294
+ '{"verdict": ...} lines on stdout. Repeat per judge.',
295
+ )
296
+ s.add_argument(
297
+ "--enable-judge-plugin",
298
+ action="store_true",
299
+ help="required: allow eval-builder to start the judge command",
300
+ )
301
+ s.add_argument("--timeout", type=float, default=120.0, help="seconds per request")
302
+ s.add_argument("--restart", action="store_true", help="discard existing judgments.jsonl")
303
+ s.set_defaults(func=_cmd_judge_run)
304
+
305
+ s = cmd("judge-check", "Measure judge stability, human agreement and bias.")
306
+ s.add_argument("--judgments", help="judgments file (default: <workspace>/judgments.jsonl)")
307
+ s.add_argument("--labels", help="human labels file (default: <workspace>/labels.jsonl)")
308
+ s.add_argument("--min-trials", type=int, default=3)
309
+ s.add_argument("--min-cases", type=int, default=10)
310
+ s.add_argument("--max-flip-rate", type=float, default=0.2)
311
+ s.add_argument("--min-position-consistency", type=float, default=0.8)
312
+ s.add_argument("--max-toward-padded", type=float, default=0.1)
313
+ s.add_argument("--min-kappa", type=float, default=0.4)
314
+ s.add_argument("--min-labeled", type=int, default=20)
315
+ s.add_argument("--full", action="store_true", help="include per-case detail in --json output")
316
+ s.set_defaults(func=_cmd_judge_check)
317
+
318
+ s = cmd("export", "Export ready cases for promptfoo, DeepEval, Inspect AI and JSONL.")
319
+ s.add_argument("--formats", help="comma separated (default: promptfoo,deepeval,inspect,jsonl)")
320
+ s.set_defaults(func=_cmd_export)
321
+
322
+ s = cmd("report", "Write report.md and report.json.")
323
+ s.add_argument("--title", help="report title")
324
+ s.set_defaults(func=_cmd_report)
325
+
326
+ s = cmd("status", "Show which steps have run.")
327
+ s.set_defaults(func=_cmd_status)
328
+
329
+ s = sub.add_parser("setup", help="Register the MCP server with Claude Code, Codex, Cursor.")
330
+ s.add_argument("--yes", action="store_true", help="apply the changes (default: show them)")
331
+ s.add_argument("--server-command", help='server command (default: "uvx eval-builder mcp")')
332
+ s.add_argument("--json", action="store_true", help="print JSON")
333
+ s.set_defaults(func=_cmd_setup)
334
+
335
+ s = sub.add_parser("mcp", help="Run the MCP server over stdio.")
336
+ s.set_defaults(func=_cmd_mcp)
337
+ return p
338
+
339
+
340
+ def main(argv: list[str] | None = None) -> int:
341
+ args = build_parser().parse_args(argv)
342
+ try:
343
+ return int(args.func(args) or 0)
344
+ except (FileNotFoundError, ValueError, KeyError) as e:
345
+ msg = e.args[0] if isinstance(e, KeyError) and e.args else str(e)
346
+ if getattr(args, "json", False):
347
+ print(json.dumps({"error": str(msg)}))
348
+ else:
349
+ print(f"error: {msg}", file=sys.stderr)
350
+ return 2
351
+
352
+
353
+ if __name__ == "__main__":
354
+ sys.exit(main())
@@ -0,0 +1,78 @@
1
+ ---
2
+ name: eval-builder
3
+ description: Turn an app's real LLM logs (OpenAI chat JSONL, Anthropic messages, Langfuse exports, OpenTelemetry GenAI spans, or input/output JSONL) into an eval suite, measure which LLM judges can be trusted (flip rate, agreement with human labels, position and length bias), and export to promptfoo, DeepEval, Inspect AI or JSONL. Use when the user wants evals from production traces, wants to know whether an LLM-as-a-judge is reliable, or wants regression tests from real conversations.
4
+ ---
5
+
6
+ # eval-builder
7
+
8
+ eval-builder is a local, deterministic tool. It never calls a model. You (the agent)
9
+ are the brain: you write expected behavior with the user, you run judges through the
10
+ user's own provider, and eval-builder does the bookkeeping, the selection and the
11
+ statistics, and records evidence for every step.
12
+
13
+ Use the MCP tools when they are available (`ingest`, `select`, `draft`, `list_cases`,
14
+ `update_case`, `set_rubric`, `validate`, `judge_plan`, `judge_check`, `export`,
15
+ `report`, `status`). Otherwise use the CLI with `--json`. All steps share one
16
+ workspace directory (default `./evalset`).
17
+
18
+ ## Workflow
19
+
20
+ 1. **Ingest.** Ask where the logs are. Run `ingest` on the files or folder. Redaction
21
+ of emails, API keys, card numbers, phone numbers and SSN-shaped numbers is on by
22
+ default. Tell the user how many traces were read, how many were skipped and why,
23
+ and what was redacted (counts by kind). Do not turn redaction off unless the user
24
+ asks.
25
+
26
+ 2. **Select.** Run `select` with `n` around 30 to 50. Pass `stratify` with the
27
+ metadata keys that matter to the user (route, feature, customer tier, model).
28
+ Failures (error flags, negative feedback) are oversampled. Show the user the
29
+ clusters and a few of the reasons cases were picked.
30
+
31
+ 3. **Draft and fill cases with the user.** Run `draft`. For each case, propose an
32
+ `expected_behavior` (one or two sentences on what a good answer does) and the
33
+ criteria it tests, then confirm with the user, especially for domain facts you are
34
+ not sure about. Use `update_case` to save it and set `status: ready`, or
35
+ `status: dropped` for cases that are not useful. Define criteria with `set_rubric`.
36
+ Never mark a case ready with a guess you have not checked; ask.
37
+
38
+ 4. **Validate.** Run `validate`. Fix every error it lists.
39
+
40
+ 5. **Plan judge runs.** Add the user's judges to the rubric (their exact judge
41
+ prompts, mode `pointwise` or `pairwise`, allowed labels). Run `judge_plan` with
42
+ `trials: 5` and `probes: ["swap", "pad"]` (swap only applies to pairwise judges).
43
+
44
+ 6. **Run the judges.** Each row of `judge_requests.jsonl` is one call, with the
45
+ rendered `prompt`. Either:
46
+ - run each request yourself through the user's provider at the temperature they
47
+ use in production, and append `{"request_id": ..., "verdict": ...}` rows to
48
+ `judgments.jsonl`; or
49
+ - if the user has a judge script, ask them to run
50
+ `eval-builder judge-run --enable-judge-plugin --command "<judge_id>=<their command>"`.
51
+ The plugin is off by default and eval-builder ships no API keys.
52
+ Do not change the prompt between trials. Repeated trials are the point.
53
+
54
+ 7. **Human labels.** Ask the user to label at least 20 cases (more is better) in
55
+ `labels.jsonl` as `{"case_id": ..., "label": ...}`. **Never write human labels
56
+ yourself and never present your own judgment as a human label.** Without labels,
57
+ no judge can be called trustworthy, and the report says so.
58
+
59
+ 8. **Check the judges.** Run `judge_check`. Each judge gets one verdict:
60
+ `trustworthy`, `unstable` (verdict flips across repeated calls), `biased` (answer
61
+ order or irrelevant padding changes the verdict), `misaligned` (low kappa with
62
+ human labels) or `not_enough_data`. Read the intervals, not only the point
63
+ estimates. Keep only trustworthy judges. For the others, report the reasons and
64
+ suggest concrete fixes: majority vote over several calls, running both answer
65
+ orders and only counting agreements, tighter rubric wording, a stronger judge model.
66
+
67
+ 9. **Export.** Run `export` with the user's tool (`promptfoo`, `deepeval`, `inspect`,
68
+ `jsonl`). The manifest lists file hashes and which judges passed.
69
+
70
+ 10. **Report.** Run `report` and give the user `report.md`. Quote numbers from it
71
+ exactly, with their intervals and sample sizes, and repeat its limits section.
72
+ Do not round a weak result into a strong claim.
73
+
74
+ ## Rules
75
+
76
+ - Everything stays on the machine. eval-builder makes no network calls.
77
+ - Every claim you make about a judge must come from `judge_check.json`.
78
+ - If a step fails, show the exact error and the command you ran.