eval-builder 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. eval_builder-0.1.0/.github/workflows/ci.yml +23 -0
  2. eval_builder-0.1.0/.github/workflows/release.yml +34 -0
  3. eval_builder-0.1.0/.gitignore +17 -0
  4. eval_builder-0.1.0/AGENTS.md +45 -0
  5. eval_builder-0.1.0/CHANGELOG.md +18 -0
  6. eval_builder-0.1.0/CONTRIBUTING.md +11 -0
  7. eval_builder-0.1.0/LICENSE +21 -0
  8. eval_builder-0.1.0/PKG-INFO +154 -0
  9. eval_builder-0.1.0/README.md +135 -0
  10. eval_builder-0.1.0/SECURITY.md +13 -0
  11. eval_builder-0.1.0/examples/judges/control_judges.py +34 -0
  12. eval_builder-0.1.0/examples/judges/ollama_judge.py +88 -0
  13. eval_builder-0.1.0/examples/mt-bench/README.md +131 -0
  14. eval_builder-0.1.0/examples/mt-bench/expected_behaviors.yaml +100 -0
  15. eval_builder-0.1.0/examples/mt-bench/fill_suite.py +91 -0
  16. eval_builder-0.1.0/examples/mt-bench/judges/cases.yaml +4620 -0
  17. eval_builder-0.1.0/examples/mt-bench/judges/judge_check.json +8902 -0
  18. eval_builder-0.1.0/examples/mt-bench/judges/judge_run_log.json +47 -0
  19. eval_builder-0.1.0/examples/mt-bench/judges/judgments.jsonl +8400 -0
  20. eval_builder-0.1.0/examples/mt-bench/judges/labels.jsonl +80 -0
  21. eval_builder-0.1.0/examples/mt-bench/judges/report.json +8969 -0
  22. eval_builder-0.1.0/examples/mt-bench/judges/report.md +44 -0
  23. eval_builder-0.1.0/examples/mt-bench/judges/rubric.yaml +35 -0
  24. eval_builder-0.1.0/examples/mt-bench/prepare.py +199 -0
  25. eval_builder-0.1.0/examples/mt-bench/run.sh +41 -0
  26. eval_builder-0.1.0/examples/mt-bench/suite/cases.yaml +930 -0
  27. eval_builder-0.1.0/examples/mt-bench/suite/exports/deepeval/dataset.json +571 -0
  28. eval_builder-0.1.0/examples/mt-bench/suite/exports/deepeval/test_eval_builder.py +57 -0
  29. eval_builder-0.1.0/examples/mt-bench/suite/exports/inspect/dataset.jsonl +24 -0
  30. eval_builder-0.1.0/examples/mt-bench/suite/exports/inspect/task.py +23 -0
  31. eval_builder-0.1.0/examples/mt-bench/suite/exports/jsonl/cases.jsonl +24 -0
  32. eval_builder-0.1.0/examples/mt-bench/suite/exports/manifest.json +51 -0
  33. eval_builder-0.1.0/examples/mt-bench/suite/exports/promptfoo/promptfooconfig.yaml +552 -0
  34. eval_builder-0.1.0/examples/mt-bench/suite/ingest.json +33 -0
  35. eval_builder-0.1.0/examples/mt-bench/suite/judge_check.json +389 -0
  36. eval_builder-0.1.0/examples/mt-bench/suite/judge_run_log.json +19 -0
  37. eval_builder-0.1.0/examples/mt-bench/suite/judgments.jsonl +240 -0
  38. eval_builder-0.1.0/examples/mt-bench/suite/report.json +1307 -0
  39. eval_builder-0.1.0/examples/mt-bench/suite/report.md +111 -0
  40. eval_builder-0.1.0/examples/mt-bench/suite/rubric.yaml +48 -0
  41. eval_builder-0.1.0/examples/mt-bench/suite/selection.json +794 -0
  42. eval_builder-0.1.0/examples/mt-bench/verify_exports.sh +19 -0
  43. eval_builder-0.1.0/pyproject.toml +52 -0
  44. eval_builder-0.1.0/skills/eval-builder/SKILL.md +78 -0
  45. eval_builder-0.1.0/src/eval_builder/__init__.py +3 -0
  46. eval_builder-0.1.0/src/eval_builder/cli.py +354 -0
  47. eval_builder-0.1.0/src/eval_builder/draft.py +271 -0
  48. eval_builder-0.1.0/src/eval_builder/export.py +284 -0
  49. eval_builder-0.1.0/src/eval_builder/ingest/__init__.py +181 -0
  50. eval_builder-0.1.0/src/eval_builder/ingest/formats.py +645 -0
  51. eval_builder-0.1.0/src/eval_builder/io.py +82 -0
  52. eval_builder-0.1.0/src/eval_builder/judge/__init__.py +1 -0
  53. eval_builder-0.1.0/src/eval_builder/judge/check.py +420 -0
  54. eval_builder-0.1.0/src/eval_builder/judge/plan.py +151 -0
  55. eval_builder-0.1.0/src/eval_builder/judge/run.py +149 -0
  56. eval_builder-0.1.0/src/eval_builder/judge/stats.py +76 -0
  57. eval_builder-0.1.0/src/eval_builder/mcp_server.py +175 -0
  58. eval_builder-0.1.0/src/eval_builder/redact.py +101 -0
  59. eval_builder-0.1.0/src/eval_builder/report.py +263 -0
  60. eval_builder-0.1.0/src/eval_builder/schema.py +242 -0
  61. eval_builder-0.1.0/src/eval_builder/select.py +412 -0
  62. eval_builder-0.1.0/src/eval_builder/setup_agents.py +190 -0
  63. eval_builder-0.1.0/src/eval_builder/status.py +32 -0
  64. eval_builder-0.1.0/src/eval_builder/workspace.py +67 -0
  65. eval_builder-0.1.0/tests/conftest.py +72 -0
  66. eval_builder-0.1.0/tests/fixtures/anthropic_messages.json +30 -0
  67. eval_builder-0.1.0/tests/fixtures/generic.jsonl +4 -0
  68. eval_builder-0.1.0/tests/fixtures/langfuse_export.json +32 -0
  69. eval_builder-0.1.0/tests/fixtures/openai_chat.jsonl +4 -0
  70. eval_builder-0.1.0/tests/fixtures/otel_genai.json +55 -0
  71. eval_builder-0.1.0/tests/test_cli_mcp.py +77 -0
  72. eval_builder-0.1.0/tests/test_draft.py +83 -0
  73. eval_builder-0.1.0/tests/test_export.py +93 -0
  74. eval_builder-0.1.0/tests/test_ingest.py +154 -0
  75. eval_builder-0.1.0/tests/test_judge_check.py +199 -0
  76. eval_builder-0.1.0/tests/test_judge_run.py +110 -0
  77. eval_builder-0.1.0/tests/test_judge_stats.py +63 -0
  78. eval_builder-0.1.0/tests/test_report.py +50 -0
  79. eval_builder-0.1.0/tests/test_select.py +171 -0
  80. eval_builder-0.1.0/tests/test_setup.py +67 -0
  81. eval_builder-0.1.0/uv.lock +1594 -0
@@ -0,0 +1,23 @@
1
+ name: ci
2
+
3
+ on:
4
+ push:
5
+ pull_request:
6
+
7
+ jobs:
8
+ test:
9
+ strategy:
10
+ fail-fast: false
11
+ matrix:
12
+ os: [ubuntu-latest, macos-latest]
13
+ python: ["3.11", "3.12", "3.13"]
14
+ runs-on: ${{ matrix.os }}
15
+ steps:
16
+ - uses: actions/checkout@v4
17
+ - uses: astral-sh/setup-uv@v6
18
+ - run: uv sync --python ${{ matrix.python }}
19
+ - run: uv run ruff check .
20
+ - run: uv run ruff format --check .
21
+ - run: uv run mypy src
22
+ - run: uv run pytest -q
23
+ - run: uv build
@@ -0,0 +1,34 @@
1
+ name: release
2
+
3
+ # Publishes to PyPI when a version tag (v*) is pushed, using PyPI trusted publishing (no API token).
4
+ on:
5
+ push:
6
+ tags: ["v*"]
7
+
8
+ permissions:
9
+ contents: read
10
+
11
+ jobs:
12
+ build:
13
+ runs-on: ubuntu-latest
14
+ steps:
15
+ - uses: actions/checkout@v4
16
+ - uses: astral-sh/setup-uv@v6
17
+ - run: uv build
18
+ - uses: actions/upload-artifact@v4
19
+ with:
20
+ name: dist
21
+ path: dist/
22
+
23
+ publish:
24
+ needs: build
25
+ runs-on: ubuntu-latest
26
+ environment: pypi
27
+ permissions:
28
+ id-token: write
29
+ steps:
30
+ - uses: actions/download-artifact@v4
31
+ with:
32
+ name: dist
33
+ path: dist/
34
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,17 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ .venv/
4
+ .pytest_cache/
5
+ .ruff_cache/
6
+ .mypy_cache/
7
+ dist/
8
+ build/
9
+ *.egg-info/
10
+ .DS_Store
11
+ evalset/
12
+ .deepeval/
13
+ # Example inputs that prepare.py downloads or derives (reproducible from the pinned dataset)
14
+ examples/mt-bench/data/
15
+ # Large or derived files inside example workspaces
16
+ examples/mt-bench/*/judge_requests.jsonl
17
+ examples/mt-bench/*/traces.jsonl
@@ -0,0 +1,45 @@
1
+ # Working on eval-builder
2
+
3
+ This file is for people and coding agents changing eval-builder itself. The agent
4
+ workflow for *using* the tool lives in `skills/eval-builder/SKILL.md`.
5
+
6
+ ## Layout
7
+
8
+ ```
9
+ src/eval_builder/
10
+ schema.py Trace dataclass and helpers every ingest format maps into
11
+ ingest/ format detection and parsers (openai, anthropic, langfuse, otel, generic)
12
+ redact.py default-on secret and PII redaction with counts
13
+ select.py dedupe, near-duplicates, k-means clusters, strata, failure oversampling
14
+ draft.py case and rubric skeletons, validation, update_case
15
+ judge/plan.py judge requests with repeated trials and swap/pad probes
16
+ judge/run.py opt-in judge plugin runner (off by default, runs a user command)
17
+ judge/check.py flip rate, kappa, accuracy, probes, verdicts
18
+ judge/stats.py Wilson interval, Cohen's kappa, majority vote
19
+ export.py promptfoo, DeepEval, Inspect AI, JSONL
20
+ report.py report.md and report.json
21
+ setup_agents.py `eval-builder setup` for Claude Code, Codex, Cursor
22
+ mcp_server.py MCP tools over stdio
23
+ cli.py argparse CLI, every command supports --json
24
+ ```
25
+
26
+ ## Rules
27
+
28
+ - No model calls and no network access in the package. The judge runner only starts a
29
+ command the user names, and only with an explicit flag.
30
+ - Deterministic: same inputs and seed give the same selection and the same files.
31
+ - Every number the tool reports must be traceable to a file in the workspace.
32
+ - Tests are offline, fast, and never touch the real home directory (`conftest.py`
33
+ points HOME at a temp dir). Add a fixture when you add a format.
34
+ - Statistics changes need a test against a value computed by hand, with the arithmetic
35
+ in a comment.
36
+ - Writing style: no em dashes or en dashes, no hype words, plain sentences.
37
+
38
+ ## Commands
39
+
40
+ ```
41
+ uv sync
42
+ uv run pytest -q
43
+ uv run ruff check . && uv run ruff format --check .
44
+ uv run mypy src
45
+ ```
@@ -0,0 +1,18 @@
1
+ # Changelog
2
+
3
+ ## 0.1.0 (2026-10-08)
4
+
5
+ First release.
6
+
7
+ - `ingest`: OpenAI chat JSONL (fine-tuning style and request/response logs), Anthropic
8
+ messages, Langfuse trace exports, OpenTelemetry GenAI spans (OTLP JSON and flat span
9
+ JSON), and generic input/output JSONL. Default-on redaction with a per-kind report.
10
+ - `select`: exact and near-duplicate removal, TF-IDF k-means clusters, stratum
11
+ coverage, failure oversampling, a recorded reason for every pick.
12
+ - `draft`, `validate`: case and rubric templates that the agent fills in with the user.
13
+ - `judge-plan`, `judge-run` (opt-in), `judge-check`: repeated trials, flip rate,
14
+ agreement with human labels (accuracy with Wilson intervals, Cohen's kappa), answer
15
+ order and padding probes, a verdict per judge.
16
+ - `export`: promptfoo, DeepEval, Inspect AI, JSONL.
17
+ - `report`: Markdown and JSON.
18
+ - `setup`: registers the MCP server with Claude Code, Codex and Cursor.
@@ -0,0 +1,11 @@
1
+ # Contributing
2
+
3
+ Thanks for helping. Before opening a pull request:
4
+
5
+ 1. `uv sync`
6
+ 2. `uv run pytest -q` (offline, under a minute)
7
+ 3. `uv run ruff check . && uv run ruff format --check . && uv run mypy src`
8
+
9
+ New log formats need a small fixture in `tests/fixtures/` and a test. Changes to the
10
+ judge statistics need a test against a hand-computed value. See `AGENTS.md` for the
11
+ layout and the rules (no model calls, deterministic output, no em dashes in text).
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Abel Yagubyan
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,154 @@
1
+ Metadata-Version: 2.5
2
+ Name: eval-builder
3
+ Version: 0.1.0
4
+ Summary: Turn real LLM app logs into an eval suite and measure which LLM judges you can trust.
5
+ Author: Abel Yagubyan
6
+ License-Expression: MIT
7
+ License-File: LICENSE
8
+ Keywords: deepeval,evals,inspect-ai,llm,llm-as-a-judge,mcp,promptfoo
9
+ Classifier: Operating System :: OS Independent
10
+ Classifier: Programming Language :: Python :: 3
11
+ Classifier: Topic :: Software Development :: Testing
12
+ Requires-Python: >=3.11
13
+ Requires-Dist: mcp>=1.2
14
+ Requires-Dist: numpy>=1.24
15
+ Requires-Dist: pyyaml>=6.0
16
+ Requires-Dist: scikit-learn>=1.3
17
+ Requires-Dist: scipy>=1.10
18
+ Description-Content-Type: text/markdown
19
+
20
+ # eval-builder
21
+
22
+ Your agent turns your app's real logs into an eval suite, and tells you which of its judges you can actually trust.
23
+
24
+ ```sh
25
+ uvx eval-builder ingest ./logs # OpenAI, Anthropic, Langfuse, OpenTelemetry or JSONL
26
+ uvx eval-builder select -n 40 && uvx eval-builder draft
27
+ uvx eval-builder judge-check # after your agent has run each judge a few times
28
+ ```
29
+
30
+ > Not on PyPI yet. Until the first release, run it straight from GitHub by replacing `uvx eval-builder` with
31
+ > `uvx --from git+https://github.com/Abelo9996/eval-builder eval-builder`. `setup` registers `uvx eval-builder mcp`, so it works once the
32
+ > package is on PyPI.
33
+
34
+ eval-builder is not another eval platform. It builds the suite and checks the judges,
35
+ then exports to the tools you already run: promptfoo, DeepEval, Inspect AI, or plain
36
+ JSONL. It never calls a model. Your coding agent (Claude Code, Codex, Cursor) does the
37
+ thinking through the MCP server; eval-builder does the selection, the bookkeeping and
38
+ the statistics, and writes down the evidence.
39
+
40
+ ## Example: real output
41
+
42
+ Everything below comes from runs on this machine (Apple M4, 16 GB) that are committed
43
+ in [`examples/mt-bench/`](examples/mt-bench/). The data is the LMSYS
44
+ [MT-Bench human judgments](https://huggingface.co/datasets/lmsys/mt_bench_human_judgments)
45
+ dataset (CC BY 4.0): 80 two-turn questions answered by 6 models, plus pairwise verdicts
46
+ from expert human judges.
47
+
48
+ **Logs to suite.** `ingest` read 480 conversations (OpenAI chat format, redaction on,
49
+ 0 matches). `select -n 24 --stratify category` removed 400 exact duplicates (each
50
+ question was asked to six models), clustered the 80 unique conversations into 9 topics
51
+ and picked 24 covering all 8 categories, 19 of them failures. From
52
+ [`suite/report.md`](examples/mt-bench/suite/report.md):
53
+
54
+ ```
55
+ | trace | cluster | represents | why it was picked |
56
+ | q81-alpaca-13b | 0 | 6 | failure (negative user feedback); 62% of unique traces are failures and at least 30% of picks are reserved for them |
57
+ | q82-alpaca-13b | 0 | 6 | adds variety within cluster 0 (11 traces, 14%; response, previous response, previous); least similar to cases already picked there |
58
+ | q156-alpaca-13b | 3 | 6 | covers category=humanities (10 unique traces, 12%) |
59
+ ```
60
+
61
+ The agent wrote expected behavior for the 24 cases, and `export` produced files that
62
+ Inspect AI 0.3.277, DeepEval 4.2.8 and promptfoo 0.124.0 all load (see
63
+ [`verify_exports.sh`](examples/mt-bench/verify_exports.sh)).
64
+
65
+ **Which judges can you trust?** Five local LLM judge setups (four models through
66
+ Ollama, one of them also at temperature 0) and two controls that are not language models judged 80 answer pairs where human experts
67
+ picked a winner. Each judge ran 5 times per pair, 5 more with the answers swapped, and
68
+ 5 more with an irrelevant paragraph appended to one answer: 8,400 calls, 0 errors.
69
+ `eval-builder judge-check`, from [`judges/report.md`](examples/mt-bench/judges/report.md):
70
+
71
+ | judge | verdict | flip rate | accuracy vs humans | kappa | survives answer swap | first-shown answer picked | padding helped |
72
+ |---|---|---|---|---|---|---|---|
73
+ | qwen2.5:7b-instruct, temp 0 | biased | 0% [0%, 5%] | 72% [62%, 81%] | 0.45 [0.26, 0.65] | 75% [65%, 83%] | 44% | 0% |
74
+ | qwen2.5:7b-instruct, temp 0.8 | biased | 12% [7%, 22%] | 70% [59%, 79%] | 0.40 [0.20, 0.60] | 75% [65%, 83%] | 42% | 1% |
75
+ | qwen2.5:3b-instruct | biased | 12% [7%, 22%] | 56% [45%, 67%] | 0.13 [-0.08, 0.35] | 44% [33%, 55%] | 35% | 9% |
76
+ | gemma2:2b | unstable | 59% [48%, 69%] | 56% [45%, 67%] | 0.14 [-0.07, 0.36] | 38% [28%, 48%] | 35% | 11% |
77
+ | llama3.2:3b | unstable | 56% [45%, 67%] | 51% [40%, 62%] | 0.04 [-0.17, 0.26] | 14% [8%, 23%] | 20% | 10% |
78
+ | control: longer answer wins | biased | 0% [0%, 5%] | 66% [55%, 76%] | 0.32 [0.12, 0.53] | 100% [95%, 100%] | 50% | 30% |
79
+ | control: coin flip | unstable | 95% [88%, 98%] | 51% [40%, 62%] | 0.03 [-0.19, 0.24] | 55% [44%, 65%] | 50% | 25% |
80
+
81
+ n = 80 pairs per judge; brackets are 95% intervals. No judge passed. The closest,
82
+ qwen2.5 7B, reaches kappa 0.40 to 0.45 with the human experts, but its verdict
83
+ flips on a quarter of the pairs when the answer order is swapped, and setting the
84
+ temperature to 0 removes the run-to-run flips without removing that order effect.
85
+ llama3.2 3B picked whichever answer it saw second 80% of the time. The longer-answer
86
+ rule never flips and agrees with humans 66% of the time, which is why stability and
87
+ accuracy alone are not enough: the padding probe catches it.
88
+
89
+ The same check on the suite's own pass/fail judge (qwen2.5 7B, 24 cases, no human
90
+ labels) gives `unstable`: its verdict changed across 5 identical calls on 7 of 24
91
+ cases (29%, interval 15% to 49%).
92
+
93
+ ## How it works
94
+
95
+ | step | what it does | what it uses |
96
+ |---|---|---|
97
+ | `ingest` | Reads OpenAI chat JSONL (fine-tuning style or request/response logs), Anthropic messages, Langfuse trace exports, OpenTelemetry GenAI spans (OTLP JSON, current `gen_ai.input.messages` and older `gen_ai.prompt.N` attributes), or generic input/output JSONL. Normalizes to one schema with input, output, earlier turns, tools called, error flag, user feedback, route and model. Redacts emails, API keys, Luhn-valid card numbers, phone numbers and SSN-shaped numbers by default and reports counts. | Python standard library |
98
+ | `select` | Removes exact duplicates (normalized hash of the user turns) and near-duplicates (character n-gram cosine), clusters the rest by topic, then picks: failures first (errors, negative feedback), at least one case per stratum (route, tool, feedback, any metadata key you name), one central case per cluster, and fills the rest by cluster size with the least similar remaining cases. Every pick records why it was picked. | scikit-learn (TF-IDF, k-means) |
99
+ | `draft`, `validate` | Writes `cases.yaml` and `rubric.yaml` with TODO markers. The agent fills in expected behavior and criteria with you. Validation refuses ready cases that still contain TODO or reference unknown criteria. | PyYAML |
100
+ | `judge-plan` | Lists every judge call to make: each case N times, plus probes that swap the answer order (pairwise judges) and pad an answer with an irrelevant paragraph. | |
101
+ | `judge-run` | Optional and off by default. Sends each request as a JSON line to a command you name (your script, your provider, your keys) and records the verdicts. eval-builder ships no API keys and no provider code. | your command |
102
+ | `judge-check` | Per judge: flip rate across repeated calls (with a Wilson interval), self-agreement, majority-of-3 vote stability, accuracy and Cohen's kappa against your human labels (with intervals), position consistency and first-shown preference, and how often padding moved the verdict toward the padded answer. Verdict: `trustworthy`, `unstable`, `biased`, `misaligned`, or `not_enough_data`, with the numbers behind it. | |
103
+ | `export` | promptfoo `promptfooconfig.yaml` (llm-rubric asserts), DeepEval dataset plus a `deepeval test run` file, Inspect AI dataset plus `task.py`, plain JSONL. The manifest lists file hashes and which judges passed. | |
104
+ | `report` | `report.md` and `report.json`: sources with sha256, counts, redactions, selection reasons, the judge table, and the limits. | |
105
+
106
+ The verdict thresholds are explicit flags with defaults: flip rate at most 20% of cases,
107
+ position consistency at least 80%, padding helps at most 10% of cases, kappa at least
108
+ 0.4 on at least 20 human-labeled cases. A judge without human labels is never called
109
+ trustworthy.
110
+
111
+ ## Setup for agents
112
+
113
+ ```sh
114
+ uvx eval-builder setup # shows what it would change
115
+ uvx eval-builder setup --yes # applies it
116
+ ```
117
+
118
+ `setup` registers the MCP server (`uvx eval-builder mcp`) with Claude Code (`claude mcp
119
+ add --scope user`), Codex (`[mcp_servers.eval-builder]` in `~/.codex/config.toml`) and
120
+ Cursor (`~/.cursor/mcp.json`), and copies the agent instructions to
121
+ `~/.claude/skills/eval-builder/` and `~/.codex/skills/eval-builder/`. It backs up any
122
+ file it edits and does nothing on a second run. The workflow the agent follows is in
123
+ [`skills/eval-builder/SKILL.md`](skills/eval-builder/SKILL.md).
124
+
125
+ MCP tools: `ingest`, `select`, `draft`, `list_cases`, `update_case`, `set_rubric`,
126
+ `validate`, `judge_plan`, `judge_check`, `export`, `report`, `status`. The judge runner
127
+ is CLI only, because it executes a command.
128
+
129
+ ## What it can't do
130
+
131
+ - It does not write expected behavior or human labels. The agent drafts expected
132
+ behavior with you; labels must come from people. Without labels, judge-check can
133
+ tell you a judge is unstable or biased, but not that it is right.
134
+ - Selection is lexical. Two requests that mean the same thing in different words can
135
+ land in different clusters, and near-duplicate detection only catches close textual
136
+ matches.
137
+ - Redaction is pattern-based. Names, street addresses and free-form secrets get
138
+ through. Look at `traces.jsonl` before sharing a workspace.
139
+ - The bias probes cover answer order and irrelevant length only. Self-preference,
140
+ style bias and rubric misreadings are not measured.
141
+ - Small samples give wide intervals. The report prints them; read them.
142
+ - It does not run your app or your eval. The exported files do that in promptfoo,
143
+ DeepEval or Inspect AI.
144
+
145
+ ## Privacy and safety
146
+
147
+ Everything runs locally. eval-builder makes no network calls and sends nothing
148
+ anywhere. The only process it starts is the judge command you name with
149
+ `judge-run --enable-judge-plugin`. Redaction is on unless you pass `--no-redact`, and
150
+ the report lists what was replaced (counts and kinds, never the values).
151
+
152
+ ## License
153
+
154
+ MIT. See [LICENSE](LICENSE).
@@ -0,0 +1,135 @@
1
+ # eval-builder
2
+
3
+ Your agent turns your app's real logs into an eval suite, and tells you which of its judges you can actually trust.
4
+
5
+ ```sh
6
+ uvx eval-builder ingest ./logs # OpenAI, Anthropic, Langfuse, OpenTelemetry or JSONL
7
+ uvx eval-builder select -n 40 && uvx eval-builder draft
8
+ uvx eval-builder judge-check # after your agent has run each judge a few times
9
+ ```
10
+
11
+ > Not on PyPI yet. Until the first release, run it straight from GitHub by replacing `uvx eval-builder` with
12
+ > `uvx --from git+https://github.com/Abelo9996/eval-builder eval-builder`. `setup` registers `uvx eval-builder mcp`, so it works once the
13
+ > package is on PyPI.
14
+
15
+ eval-builder is not another eval platform. It builds the suite and checks the judges,
16
+ then exports to the tools you already run: promptfoo, DeepEval, Inspect AI, or plain
17
+ JSONL. It never calls a model. Your coding agent (Claude Code, Codex, Cursor) does the
18
+ thinking through the MCP server; eval-builder does the selection, the bookkeeping and
19
+ the statistics, and writes down the evidence.
20
+
21
+ ## Example: real output
22
+
23
+ Everything below comes from runs on this machine (Apple M4, 16 GB) that are committed
24
+ in [`examples/mt-bench/`](examples/mt-bench/). The data is the LMSYS
25
+ [MT-Bench human judgments](https://huggingface.co/datasets/lmsys/mt_bench_human_judgments)
26
+ dataset (CC BY 4.0): 80 two-turn questions answered by 6 models, plus pairwise verdicts
27
+ from expert human judges.
28
+
29
+ **Logs to suite.** `ingest` read 480 conversations (OpenAI chat format, redaction on,
30
+ 0 matches). `select -n 24 --stratify category` removed 400 exact duplicates (each
31
+ question was asked to six models), clustered the 80 unique conversations into 9 topics
32
+ and picked 24 covering all 8 categories, 19 of them failures. From
33
+ [`suite/report.md`](examples/mt-bench/suite/report.md):
34
+
35
+ ```
36
+ | trace | cluster | represents | why it was picked |
37
+ | q81-alpaca-13b | 0 | 6 | failure (negative user feedback); 62% of unique traces are failures and at least 30% of picks are reserved for them |
38
+ | q82-alpaca-13b | 0 | 6 | adds variety within cluster 0 (11 traces, 14%; response, previous response, previous); least similar to cases already picked there |
39
+ | q156-alpaca-13b | 3 | 6 | covers category=humanities (10 unique traces, 12%) |
40
+ ```
41
+
42
+ The agent wrote expected behavior for the 24 cases, and `export` produced files that
43
+ Inspect AI 0.3.277, DeepEval 4.2.8 and promptfoo 0.124.0 all load (see
44
+ [`verify_exports.sh`](examples/mt-bench/verify_exports.sh)).
45
+
46
+ **Which judges can you trust?** Five local LLM judge setups (four models through
47
+ Ollama, one of them also at temperature 0) and two controls that are not language models judged 80 answer pairs where human experts
48
+ picked a winner. Each judge ran 5 times per pair, 5 more with the answers swapped, and
49
+ 5 more with an irrelevant paragraph appended to one answer: 8,400 calls, 0 errors.
50
+ `eval-builder judge-check`, from [`judges/report.md`](examples/mt-bench/judges/report.md):
51
+
52
+ | judge | verdict | flip rate | accuracy vs humans | kappa | survives answer swap | first-shown answer picked | padding helped |
53
+ |---|---|---|---|---|---|---|---|
54
+ | qwen2.5:7b-instruct, temp 0 | biased | 0% [0%, 5%] | 72% [62%, 81%] | 0.45 [0.26, 0.65] | 75% [65%, 83%] | 44% | 0% |
55
+ | qwen2.5:7b-instruct, temp 0.8 | biased | 12% [7%, 22%] | 70% [59%, 79%] | 0.40 [0.20, 0.60] | 75% [65%, 83%] | 42% | 1% |
56
+ | qwen2.5:3b-instruct | biased | 12% [7%, 22%] | 56% [45%, 67%] | 0.13 [-0.08, 0.35] | 44% [33%, 55%] | 35% | 9% |
57
+ | gemma2:2b | unstable | 59% [48%, 69%] | 56% [45%, 67%] | 0.14 [-0.07, 0.36] | 38% [28%, 48%] | 35% | 11% |
58
+ | llama3.2:3b | unstable | 56% [45%, 67%] | 51% [40%, 62%] | 0.04 [-0.17, 0.26] | 14% [8%, 23%] | 20% | 10% |
59
+ | control: longer answer wins | biased | 0% [0%, 5%] | 66% [55%, 76%] | 0.32 [0.12, 0.53] | 100% [95%, 100%] | 50% | 30% |
60
+ | control: coin flip | unstable | 95% [88%, 98%] | 51% [40%, 62%] | 0.03 [-0.19, 0.24] | 55% [44%, 65%] | 50% | 25% |
61
+
62
+ n = 80 pairs per judge; brackets are 95% intervals. No judge passed. The closest,
63
+ qwen2.5 7B, reaches kappa 0.40 to 0.45 with the human experts, but its verdict
64
+ flips on a quarter of the pairs when the answer order is swapped, and setting the
65
+ temperature to 0 removes the run-to-run flips without removing that order effect.
66
+ llama3.2 3B picked whichever answer it saw second 80% of the time. The longer-answer
67
+ rule never flips and agrees with humans 66% of the time, which is why stability and
68
+ accuracy alone are not enough: the padding probe catches it.
69
+
70
+ The same check on the suite's own pass/fail judge (qwen2.5 7B, 24 cases, no human
71
+ labels) gives `unstable`: its verdict changed across 5 identical calls on 7 of 24
72
+ cases (29%, interval 15% to 49%).
73
+
74
+ ## How it works
75
+
76
+ | step | what it does | what it uses |
77
+ |---|---|---|
78
+ | `ingest` | Reads OpenAI chat JSONL (fine-tuning style or request/response logs), Anthropic messages, Langfuse trace exports, OpenTelemetry GenAI spans (OTLP JSON, current `gen_ai.input.messages` and older `gen_ai.prompt.N` attributes), or generic input/output JSONL. Normalizes to one schema with input, output, earlier turns, tools called, error flag, user feedback, route and model. Redacts emails, API keys, Luhn-valid card numbers, phone numbers and SSN-shaped numbers by default and reports counts. | Python standard library |
79
+ | `select` | Removes exact duplicates (normalized hash of the user turns) and near-duplicates (character n-gram cosine), clusters the rest by topic, then picks: failures first (errors, negative feedback), at least one case per stratum (route, tool, feedback, any metadata key you name), one central case per cluster, and fills the rest by cluster size with the least similar remaining cases. Every pick records why it was picked. | scikit-learn (TF-IDF, k-means) |
80
+ | `draft`, `validate` | Writes `cases.yaml` and `rubric.yaml` with TODO markers. The agent fills in expected behavior and criteria with you. Validation refuses ready cases that still contain TODO or reference unknown criteria. | PyYAML |
81
+ | `judge-plan` | Lists every judge call to make: each case N times, plus probes that swap the answer order (pairwise judges) and pad an answer with an irrelevant paragraph. | |
82
+ | `judge-run` | Optional and off by default. Sends each request as a JSON line to a command you name (your script, your provider, your keys) and records the verdicts. eval-builder ships no API keys and no provider code. | your command |
83
+ | `judge-check` | Per judge: flip rate across repeated calls (with a Wilson interval), self-agreement, majority-of-3 vote stability, accuracy and Cohen's kappa against your human labels (with intervals), position consistency and first-shown preference, and how often padding moved the verdict toward the padded answer. Verdict: `trustworthy`, `unstable`, `biased`, `misaligned`, or `not_enough_data`, with the numbers behind it. | |
84
+ | `export` | promptfoo `promptfooconfig.yaml` (llm-rubric asserts), DeepEval dataset plus a `deepeval test run` file, Inspect AI dataset plus `task.py`, plain JSONL. The manifest lists file hashes and which judges passed. | |
85
+ | `report` | `report.md` and `report.json`: sources with sha256, counts, redactions, selection reasons, the judge table, and the limits. | |
86
+
87
+ The verdict thresholds are explicit flags with defaults: flip rate at most 20% of cases,
88
+ position consistency at least 80%, padding helps at most 10% of cases, kappa at least
89
+ 0.4 on at least 20 human-labeled cases. A judge without human labels is never called
90
+ trustworthy.
91
+
92
+ ## Setup for agents
93
+
94
+ ```sh
95
+ uvx eval-builder setup # shows what it would change
96
+ uvx eval-builder setup --yes # applies it
97
+ ```
98
+
99
+ `setup` registers the MCP server (`uvx eval-builder mcp`) with Claude Code (`claude mcp
100
+ add --scope user`), Codex (`[mcp_servers.eval-builder]` in `~/.codex/config.toml`) and
101
+ Cursor (`~/.cursor/mcp.json`), and copies the agent instructions to
102
+ `~/.claude/skills/eval-builder/` and `~/.codex/skills/eval-builder/`. It backs up any
103
+ file it edits and does nothing on a second run. The workflow the agent follows is in
104
+ [`skills/eval-builder/SKILL.md`](skills/eval-builder/SKILL.md).
105
+
106
+ MCP tools: `ingest`, `select`, `draft`, `list_cases`, `update_case`, `set_rubric`,
107
+ `validate`, `judge_plan`, `judge_check`, `export`, `report`, `status`. The judge runner
108
+ is CLI only, because it executes a command.
109
+
110
+ ## What it can't do
111
+
112
+ - It does not write expected behavior or human labels. The agent drafts expected
113
+ behavior with you; labels must come from people. Without labels, judge-check can
114
+ tell you a judge is unstable or biased, but not that it is right.
115
+ - Selection is lexical. Two requests that mean the same thing in different words can
116
+ land in different clusters, and near-duplicate detection only catches close textual
117
+ matches.
118
+ - Redaction is pattern-based. Names, street addresses and free-form secrets get
119
+ through. Look at `traces.jsonl` before sharing a workspace.
120
+ - The bias probes cover answer order and irrelevant length only. Self-preference,
121
+ style bias and rubric misreadings are not measured.
122
+ - Small samples give wide intervals. The report prints them; read them.
123
+ - It does not run your app or your eval. The exported files do that in promptfoo,
124
+ DeepEval or Inspect AI.
125
+
126
+ ## Privacy and safety
127
+
128
+ Everything runs locally. eval-builder makes no network calls and sends nothing
129
+ anywhere. The only process it starts is the judge command you name with
130
+ `judge-run --enable-judge-plugin`. Redaction is on unless you pass `--no-redact`, and
131
+ the report lists what was replaced (counts and kinds, never the values).
132
+
133
+ ## License
134
+
135
+ MIT. See [LICENSE](LICENSE).
@@ -0,0 +1,13 @@
1
+ # Security
2
+
3
+ eval-builder reads log files you point it at and writes files into a workspace
4
+ directory. It makes no network calls. The optional judge runner starts only the
5
+ command you pass with `--command`, and only when you also pass
6
+ `--enable-judge-plugin`.
7
+
8
+ Logs often contain personal data. Redaction is on by default but is pattern-based and
9
+ will miss things (names, addresses, free-form secrets). Review the workspace before
10
+ sharing it.
11
+
12
+ To report a vulnerability, open a private security advisory on the GitHub repository
13
+ or email the maintainer. Please do not file a public issue for security problems.
@@ -0,0 +1,34 @@
1
+ """Control judges for `eval-builder judge-run`. These are NOT language models.
2
+
3
+ --kind longer picks whichever answer is longer (a deterministic heuristic)
4
+ --kind coin picks A or B at random (seeded by request id), a pure-noise control
5
+
6
+ They exist to show that judge-check separates "stable but biased" (longer) and
7
+ "noise" (coin) from real judges. Never report them as LLM judges.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import argparse
13
+ import hashlib
14
+ import json
15
+ import sys
16
+
17
+
18
+ def main() -> None:
19
+ ap = argparse.ArgumentParser()
20
+ ap.add_argument("--kind", choices=["longer", "coin"], required=True)
21
+ args = ap.parse_args()
22
+ for line in sys.stdin:
23
+ req = json.loads(line)
24
+ shown = req["presented"]
25
+ if args.kind == "longer":
26
+ verdict = "A" if len(shown["answer_a"]) >= len(shown["answer_b"]) else "B"
27
+ else:
28
+ h = hashlib.sha256(req["request_id"].encode()).digest()[0]
29
+ verdict = "A" if h % 2 == 0 else "B"
30
+ print(json.dumps({"verdict": verdict}), flush=True)
31
+
32
+
33
+ if __name__ == "__main__":
34
+ main()
@@ -0,0 +1,88 @@
1
+ """Example judge plugin for `eval-builder judge-run`: a local model served by Ollama.
2
+
3
+ This is an example, not part of the eval-builder package. It talks only to the
4
+ Ollama server on localhost. Protocol: read one JSON request per line on stdin,
5
+ write one JSON response per line on stdout: {"verdict": ..., "raw": ...}.
6
+
7
+ Usage:
8
+ eval-builder judge-run --enable-judge-plugin \\
9
+ --command "qwen-3b=python examples/judges/ollama_judge.py --model qwen2.5:3b-instruct"
10
+
11
+ Each trial uses seed = trial index, so a rerun on the same machine, model digest and
12
+ Ollama version reproduces the same outputs.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import argparse
18
+ import json
19
+ import sys
20
+ import urllib.request
21
+
22
+
23
+ def generate(
24
+ host: str,
25
+ model: str,
26
+ prompt: str,
27
+ temperature: float,
28
+ seed: int,
29
+ num_ctx: int,
30
+ num_predict: int,
31
+ ) -> dict:
32
+ body = json.dumps(
33
+ {
34
+ "model": model,
35
+ "prompt": prompt,
36
+ "stream": False,
37
+ "options": {
38
+ "temperature": temperature,
39
+ "seed": seed,
40
+ "num_ctx": num_ctx,
41
+ "num_predict": num_predict,
42
+ },
43
+ }
44
+ ).encode()
45
+ req = urllib.request.Request(f"{host}/api/generate", body, {"Content-Type": "application/json"})
46
+ with urllib.request.urlopen(req, timeout=300) as resp:
47
+ return json.load(resp)
48
+
49
+
50
+ def main() -> None:
51
+ ap = argparse.ArgumentParser()
52
+ ap.add_argument("--model", required=True)
53
+ ap.add_argument("--temperature", type=float, default=0.8)
54
+ ap.add_argument("--num-ctx", type=int, default=8192)
55
+ ap.add_argument("--num-predict", type=int, default=8)
56
+ ap.add_argument("--host", default="http://127.0.0.1:11434")
57
+ args = ap.parse_args()
58
+ for line in sys.stdin:
59
+ req = json.loads(line)
60
+ if not req.get("prompt"):
61
+ print(
62
+ json.dumps({"verdict": None, "raw": "request has no rendered prompt"}), flush=True
63
+ )
64
+ continue
65
+ out = generate(
66
+ args.host,
67
+ args.model,
68
+ req["prompt"],
69
+ args.temperature,
70
+ int(req.get("trial", 0)),
71
+ args.num_ctx,
72
+ args.num_predict,
73
+ )
74
+ text = out.get("response", "")
75
+ print(
76
+ json.dumps(
77
+ {
78
+ "verdict": text.strip(),
79
+ "raw": text,
80
+ "prompt_tokens": out.get("prompt_eval_count"),
81
+ }
82
+ ),
83
+ flush=True,
84
+ )
85
+
86
+
87
+ if __name__ == "__main__":
88
+ main()