eval-builder 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- eval_builder-0.1.0/.github/workflows/ci.yml +23 -0
- eval_builder-0.1.0/.github/workflows/release.yml +34 -0
- eval_builder-0.1.0/.gitignore +17 -0
- eval_builder-0.1.0/AGENTS.md +45 -0
- eval_builder-0.1.0/CHANGELOG.md +18 -0
- eval_builder-0.1.0/CONTRIBUTING.md +11 -0
- eval_builder-0.1.0/LICENSE +21 -0
- eval_builder-0.1.0/PKG-INFO +154 -0
- eval_builder-0.1.0/README.md +135 -0
- eval_builder-0.1.0/SECURITY.md +13 -0
- eval_builder-0.1.0/examples/judges/control_judges.py +34 -0
- eval_builder-0.1.0/examples/judges/ollama_judge.py +88 -0
- eval_builder-0.1.0/examples/mt-bench/README.md +131 -0
- eval_builder-0.1.0/examples/mt-bench/expected_behaviors.yaml +100 -0
- eval_builder-0.1.0/examples/mt-bench/fill_suite.py +91 -0
- eval_builder-0.1.0/examples/mt-bench/judges/cases.yaml +4620 -0
- eval_builder-0.1.0/examples/mt-bench/judges/judge_check.json +8902 -0
- eval_builder-0.1.0/examples/mt-bench/judges/judge_run_log.json +47 -0
- eval_builder-0.1.0/examples/mt-bench/judges/judgments.jsonl +8400 -0
- eval_builder-0.1.0/examples/mt-bench/judges/labels.jsonl +80 -0
- eval_builder-0.1.0/examples/mt-bench/judges/report.json +8969 -0
- eval_builder-0.1.0/examples/mt-bench/judges/report.md +44 -0
- eval_builder-0.1.0/examples/mt-bench/judges/rubric.yaml +35 -0
- eval_builder-0.1.0/examples/mt-bench/prepare.py +199 -0
- eval_builder-0.1.0/examples/mt-bench/run.sh +41 -0
- eval_builder-0.1.0/examples/mt-bench/suite/cases.yaml +930 -0
- eval_builder-0.1.0/examples/mt-bench/suite/exports/deepeval/dataset.json +571 -0
- eval_builder-0.1.0/examples/mt-bench/suite/exports/deepeval/test_eval_builder.py +57 -0
- eval_builder-0.1.0/examples/mt-bench/suite/exports/inspect/dataset.jsonl +24 -0
- eval_builder-0.1.0/examples/mt-bench/suite/exports/inspect/task.py +23 -0
- eval_builder-0.1.0/examples/mt-bench/suite/exports/jsonl/cases.jsonl +24 -0
- eval_builder-0.1.0/examples/mt-bench/suite/exports/manifest.json +51 -0
- eval_builder-0.1.0/examples/mt-bench/suite/exports/promptfoo/promptfooconfig.yaml +552 -0
- eval_builder-0.1.0/examples/mt-bench/suite/ingest.json +33 -0
- eval_builder-0.1.0/examples/mt-bench/suite/judge_check.json +389 -0
- eval_builder-0.1.0/examples/mt-bench/suite/judge_run_log.json +19 -0
- eval_builder-0.1.0/examples/mt-bench/suite/judgments.jsonl +240 -0
- eval_builder-0.1.0/examples/mt-bench/suite/report.json +1307 -0
- eval_builder-0.1.0/examples/mt-bench/suite/report.md +111 -0
- eval_builder-0.1.0/examples/mt-bench/suite/rubric.yaml +48 -0
- eval_builder-0.1.0/examples/mt-bench/suite/selection.json +794 -0
- eval_builder-0.1.0/examples/mt-bench/verify_exports.sh +19 -0
- eval_builder-0.1.0/pyproject.toml +52 -0
- eval_builder-0.1.0/skills/eval-builder/SKILL.md +78 -0
- eval_builder-0.1.0/src/eval_builder/__init__.py +3 -0
- eval_builder-0.1.0/src/eval_builder/cli.py +354 -0
- eval_builder-0.1.0/src/eval_builder/draft.py +271 -0
- eval_builder-0.1.0/src/eval_builder/export.py +284 -0
- eval_builder-0.1.0/src/eval_builder/ingest/__init__.py +181 -0
- eval_builder-0.1.0/src/eval_builder/ingest/formats.py +645 -0
- eval_builder-0.1.0/src/eval_builder/io.py +82 -0
- eval_builder-0.1.0/src/eval_builder/judge/__init__.py +1 -0
- eval_builder-0.1.0/src/eval_builder/judge/check.py +420 -0
- eval_builder-0.1.0/src/eval_builder/judge/plan.py +151 -0
- eval_builder-0.1.0/src/eval_builder/judge/run.py +149 -0
- eval_builder-0.1.0/src/eval_builder/judge/stats.py +76 -0
- eval_builder-0.1.0/src/eval_builder/mcp_server.py +175 -0
- eval_builder-0.1.0/src/eval_builder/redact.py +101 -0
- eval_builder-0.1.0/src/eval_builder/report.py +263 -0
- eval_builder-0.1.0/src/eval_builder/schema.py +242 -0
- eval_builder-0.1.0/src/eval_builder/select.py +412 -0
- eval_builder-0.1.0/src/eval_builder/setup_agents.py +190 -0
- eval_builder-0.1.0/src/eval_builder/status.py +32 -0
- eval_builder-0.1.0/src/eval_builder/workspace.py +67 -0
- eval_builder-0.1.0/tests/conftest.py +72 -0
- eval_builder-0.1.0/tests/fixtures/anthropic_messages.json +30 -0
- eval_builder-0.1.0/tests/fixtures/generic.jsonl +4 -0
- eval_builder-0.1.0/tests/fixtures/langfuse_export.json +32 -0
- eval_builder-0.1.0/tests/fixtures/openai_chat.jsonl +4 -0
- eval_builder-0.1.0/tests/fixtures/otel_genai.json +55 -0
- eval_builder-0.1.0/tests/test_cli_mcp.py +77 -0
- eval_builder-0.1.0/tests/test_draft.py +83 -0
- eval_builder-0.1.0/tests/test_export.py +93 -0
- eval_builder-0.1.0/tests/test_ingest.py +154 -0
- eval_builder-0.1.0/tests/test_judge_check.py +199 -0
- eval_builder-0.1.0/tests/test_judge_run.py +110 -0
- eval_builder-0.1.0/tests/test_judge_stats.py +63 -0
- eval_builder-0.1.0/tests/test_report.py +50 -0
- eval_builder-0.1.0/tests/test_select.py +171 -0
- eval_builder-0.1.0/tests/test_setup.py +67 -0
- eval_builder-0.1.0/uv.lock +1594 -0
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
name: ci
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
pull_request:
|
|
6
|
+
|
|
7
|
+
jobs:
|
|
8
|
+
test:
|
|
9
|
+
strategy:
|
|
10
|
+
fail-fast: false
|
|
11
|
+
matrix:
|
|
12
|
+
os: [ubuntu-latest, macos-latest]
|
|
13
|
+
python: ["3.11", "3.12", "3.13"]
|
|
14
|
+
runs-on: ${{ matrix.os }}
|
|
15
|
+
steps:
|
|
16
|
+
- uses: actions/checkout@v4
|
|
17
|
+
- uses: astral-sh/setup-uv@v6
|
|
18
|
+
- run: uv sync --python ${{ matrix.python }}
|
|
19
|
+
- run: uv run ruff check .
|
|
20
|
+
- run: uv run ruff format --check .
|
|
21
|
+
- run: uv run mypy src
|
|
22
|
+
- run: uv run pytest -q
|
|
23
|
+
- run: uv build
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
name: release
|
|
2
|
+
|
|
3
|
+
# Publishes to PyPI when a version tag (v*) is pushed, using PyPI trusted publishing (no API token).
|
|
4
|
+
on:
|
|
5
|
+
push:
|
|
6
|
+
tags: ["v*"]
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: read
|
|
10
|
+
|
|
11
|
+
jobs:
|
|
12
|
+
build:
|
|
13
|
+
runs-on: ubuntu-latest
|
|
14
|
+
steps:
|
|
15
|
+
- uses: actions/checkout@v4
|
|
16
|
+
- uses: astral-sh/setup-uv@v6
|
|
17
|
+
- run: uv build
|
|
18
|
+
- uses: actions/upload-artifact@v4
|
|
19
|
+
with:
|
|
20
|
+
name: dist
|
|
21
|
+
path: dist/
|
|
22
|
+
|
|
23
|
+
publish:
|
|
24
|
+
needs: build
|
|
25
|
+
runs-on: ubuntu-latest
|
|
26
|
+
environment: pypi
|
|
27
|
+
permissions:
|
|
28
|
+
id-token: write
|
|
29
|
+
steps:
|
|
30
|
+
- uses: actions/download-artifact@v4
|
|
31
|
+
with:
|
|
32
|
+
name: dist
|
|
33
|
+
path: dist/
|
|
34
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
__pycache__/
|
|
2
|
+
*.py[cod]
|
|
3
|
+
.venv/
|
|
4
|
+
.pytest_cache/
|
|
5
|
+
.ruff_cache/
|
|
6
|
+
.mypy_cache/
|
|
7
|
+
dist/
|
|
8
|
+
build/
|
|
9
|
+
*.egg-info/
|
|
10
|
+
.DS_Store
|
|
11
|
+
evalset/
|
|
12
|
+
.deepeval/
|
|
13
|
+
# Example inputs that prepare.py downloads or derives (reproducible from the pinned dataset)
|
|
14
|
+
examples/mt-bench/data/
|
|
15
|
+
# Large or derived files inside example workspaces
|
|
16
|
+
examples/mt-bench/*/judge_requests.jsonl
|
|
17
|
+
examples/mt-bench/*/traces.jsonl
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
# Working on eval-builder
|
|
2
|
+
|
|
3
|
+
This file is for people and coding agents changing eval-builder itself. The agent
|
|
4
|
+
workflow for *using* the tool lives in `skills/eval-builder/SKILL.md`.
|
|
5
|
+
|
|
6
|
+
## Layout
|
|
7
|
+
|
|
8
|
+
```
|
|
9
|
+
src/eval_builder/
|
|
10
|
+
schema.py Trace dataclass and helpers every ingest format maps into
|
|
11
|
+
ingest/ format detection and parsers (openai, anthropic, langfuse, otel, generic)
|
|
12
|
+
redact.py default-on secret and PII redaction with counts
|
|
13
|
+
select.py dedupe, near-duplicates, k-means clusters, strata, failure oversampling
|
|
14
|
+
draft.py case and rubric skeletons, validation, update_case
|
|
15
|
+
judge/plan.py judge requests with repeated trials and swap/pad probes
|
|
16
|
+
judge/run.py opt-in judge plugin runner (off by default, runs a user command)
|
|
17
|
+
judge/check.py flip rate, kappa, accuracy, probes, verdicts
|
|
18
|
+
judge/stats.py Wilson interval, Cohen's kappa, majority vote
|
|
19
|
+
export.py promptfoo, DeepEval, Inspect AI, JSONL
|
|
20
|
+
report.py report.md and report.json
|
|
21
|
+
setup_agents.py `eval-builder setup` for Claude Code, Codex, Cursor
|
|
22
|
+
mcp_server.py MCP tools over stdio
|
|
23
|
+
cli.py argparse CLI, every command supports --json
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
## Rules
|
|
27
|
+
|
|
28
|
+
- No model calls and no network access in the package. The judge runner only starts a
|
|
29
|
+
command the user names, and only with an explicit flag.
|
|
30
|
+
- Deterministic: same inputs and seed give the same selection and the same files.
|
|
31
|
+
- Every number the tool reports must be traceable to a file in the workspace.
|
|
32
|
+
- Tests are offline, fast, and never touch the real home directory (`conftest.py`
|
|
33
|
+
points HOME at a temp dir). Add a fixture when you add a format.
|
|
34
|
+
- Statistics changes need a test against a value computed by hand, with the arithmetic
|
|
35
|
+
in a comment.
|
|
36
|
+
- Writing style: no em dashes or en dashes, no hype words, plain sentences.
|
|
37
|
+
|
|
38
|
+
## Commands
|
|
39
|
+
|
|
40
|
+
```
|
|
41
|
+
uv sync
|
|
42
|
+
uv run pytest -q
|
|
43
|
+
uv run ruff check . && uv run ruff format --check .
|
|
44
|
+
uv run mypy src
|
|
45
|
+
```
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 0.1.0 (2026-10-08)
|
|
4
|
+
|
|
5
|
+
First release.
|
|
6
|
+
|
|
7
|
+
- `ingest`: OpenAI chat JSONL (fine-tuning style and request/response logs), Anthropic
|
|
8
|
+
messages, Langfuse trace exports, OpenTelemetry GenAI spans (OTLP JSON and flat span
|
|
9
|
+
JSON), and generic input/output JSONL. Default-on redaction with a per-kind report.
|
|
10
|
+
- `select`: exact and near-duplicate removal, TF-IDF k-means clusters, stratum
|
|
11
|
+
coverage, failure oversampling, a recorded reason for every pick.
|
|
12
|
+
- `draft`, `validate`: case and rubric templates that the agent fills in with the user.
|
|
13
|
+
- `judge-plan`, `judge-run` (opt-in), `judge-check`: repeated trials, flip rate,
|
|
14
|
+
agreement with human labels (accuracy with Wilson intervals, Cohen's kappa), answer
|
|
15
|
+
order and padding probes, a verdict per judge.
|
|
16
|
+
- `export`: promptfoo, DeepEval, Inspect AI, JSONL.
|
|
17
|
+
- `report`: Markdown and JSON.
|
|
18
|
+
- `setup`: registers the MCP server with Claude Code, Codex and Cursor.
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
# Contributing
|
|
2
|
+
|
|
3
|
+
Thanks for helping. Before opening a pull request:
|
|
4
|
+
|
|
5
|
+
1. `uv sync`
|
|
6
|
+
2. `uv run pytest -q` (offline, under a minute)
|
|
7
|
+
3. `uv run ruff check . && uv run ruff format --check . && uv run mypy src`
|
|
8
|
+
|
|
9
|
+
New log formats need a small fixture in `tests/fixtures/` and a test. Changes to the
|
|
10
|
+
judge statistics need a test against a hand-computed value. See `AGENTS.md` for the
|
|
11
|
+
layout and the rules (no model calls, deterministic output, no em dashes in text).
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Abel Yagubyan
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: eval-builder
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Turn real LLM app logs into an eval suite and measure which LLM judges you can trust.
|
|
5
|
+
Author: Abel Yagubyan
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
License-File: LICENSE
|
|
8
|
+
Keywords: deepeval,evals,inspect-ai,llm,llm-as-a-judge,mcp,promptfoo
|
|
9
|
+
Classifier: Operating System :: OS Independent
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Topic :: Software Development :: Testing
|
|
12
|
+
Requires-Python: >=3.11
|
|
13
|
+
Requires-Dist: mcp>=1.2
|
|
14
|
+
Requires-Dist: numpy>=1.24
|
|
15
|
+
Requires-Dist: pyyaml>=6.0
|
|
16
|
+
Requires-Dist: scikit-learn>=1.3
|
|
17
|
+
Requires-Dist: scipy>=1.10
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
|
|
20
|
+
# eval-builder
|
|
21
|
+
|
|
22
|
+
Your agent turns your app's real logs into an eval suite, and tells you which of its judges you can actually trust.
|
|
23
|
+
|
|
24
|
+
```sh
|
|
25
|
+
uvx eval-builder ingest ./logs # OpenAI, Anthropic, Langfuse, OpenTelemetry or JSONL
|
|
26
|
+
uvx eval-builder select -n 40 && uvx eval-builder draft
|
|
27
|
+
uvx eval-builder judge-check # after your agent has run each judge a few times
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
> Not on PyPI yet. Until the first release, run it straight from GitHub by replacing `uvx eval-builder` with
|
|
31
|
+
> `uvx --from git+https://github.com/Abelo9996/eval-builder eval-builder`. `setup` registers `uvx eval-builder mcp`, so it works once the
|
|
32
|
+
> package is on PyPI.
|
|
33
|
+
|
|
34
|
+
eval-builder is not another eval platform. It builds the suite and checks the judges,
|
|
35
|
+
then exports to the tools you already run: promptfoo, DeepEval, Inspect AI, or plain
|
|
36
|
+
JSONL. It never calls a model. Your coding agent (Claude Code, Codex, Cursor) does the
|
|
37
|
+
thinking through the MCP server; eval-builder does the selection, the bookkeeping and
|
|
38
|
+
the statistics, and writes down the evidence.
|
|
39
|
+
|
|
40
|
+
## Example: real output
|
|
41
|
+
|
|
42
|
+
Everything below comes from runs on this machine (Apple M4, 16 GB) that are committed
|
|
43
|
+
in [`examples/mt-bench/`](examples/mt-bench/). The data is the LMSYS
|
|
44
|
+
[MT-Bench human judgments](https://huggingface.co/datasets/lmsys/mt_bench_human_judgments)
|
|
45
|
+
dataset (CC BY 4.0): 80 two-turn questions answered by 6 models, plus pairwise verdicts
|
|
46
|
+
from expert human judges.
|
|
47
|
+
|
|
48
|
+
**Logs to suite.** `ingest` read 480 conversations (OpenAI chat format, redaction on,
|
|
49
|
+
0 matches). `select -n 24 --stratify category` removed 400 exact duplicates (each
|
|
50
|
+
question was asked to six models), clustered the 80 unique conversations into 9 topics
|
|
51
|
+
and picked 24 covering all 8 categories, 19 of them failures. From
|
|
52
|
+
[`suite/report.md`](examples/mt-bench/suite/report.md):
|
|
53
|
+
|
|
54
|
+
```
|
|
55
|
+
| trace | cluster | represents | why it was picked |
|
|
56
|
+
| q81-alpaca-13b | 0 | 6 | failure (negative user feedback); 62% of unique traces are failures and at least 30% of picks are reserved for them |
|
|
57
|
+
| q82-alpaca-13b | 0 | 6 | adds variety within cluster 0 (11 traces, 14%; response, previous response, previous); least similar to cases already picked there |
|
|
58
|
+
| q156-alpaca-13b | 3 | 6 | covers category=humanities (10 unique traces, 12%) |
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
The agent wrote expected behavior for the 24 cases, and `export` produced files that
|
|
62
|
+
Inspect AI 0.3.277, DeepEval 4.2.8 and promptfoo 0.124.0 all load (see
|
|
63
|
+
[`verify_exports.sh`](examples/mt-bench/verify_exports.sh)).
|
|
64
|
+
|
|
65
|
+
**Which judges can you trust?** Five local LLM judge setups (four models through
|
|
66
|
+
Ollama, one of them also at temperature 0) and two controls that are not language models judged 80 answer pairs where human experts
|
|
67
|
+
picked a winner. Each judge ran 5 times per pair, 5 more with the answers swapped, and
|
|
68
|
+
5 more with an irrelevant paragraph appended to one answer: 8,400 calls, 0 errors.
|
|
69
|
+
`eval-builder judge-check`, from [`judges/report.md`](examples/mt-bench/judges/report.md):
|
|
70
|
+
|
|
71
|
+
| judge | verdict | flip rate | accuracy vs humans | kappa | survives answer swap | first-shown answer picked | padding helped |
|
|
72
|
+
|---|---|---|---|---|---|---|---|
|
|
73
|
+
| qwen2.5:7b-instruct, temp 0 | biased | 0% [0%, 5%] | 72% [62%, 81%] | 0.45 [0.26, 0.65] | 75% [65%, 83%] | 44% | 0% |
|
|
74
|
+
| qwen2.5:7b-instruct, temp 0.8 | biased | 12% [7%, 22%] | 70% [59%, 79%] | 0.40 [0.20, 0.60] | 75% [65%, 83%] | 42% | 1% |
|
|
75
|
+
| qwen2.5:3b-instruct | biased | 12% [7%, 22%] | 56% [45%, 67%] | 0.13 [-0.08, 0.35] | 44% [33%, 55%] | 35% | 9% |
|
|
76
|
+
| gemma2:2b | unstable | 59% [48%, 69%] | 56% [45%, 67%] | 0.14 [-0.07, 0.36] | 38% [28%, 48%] | 35% | 11% |
|
|
77
|
+
| llama3.2:3b | unstable | 56% [45%, 67%] | 51% [40%, 62%] | 0.04 [-0.17, 0.26] | 14% [8%, 23%] | 20% | 10% |
|
|
78
|
+
| control: longer answer wins | biased | 0% [0%, 5%] | 66% [55%, 76%] | 0.32 [0.12, 0.53] | 100% [95%, 100%] | 50% | 30% |
|
|
79
|
+
| control: coin flip | unstable | 95% [88%, 98%] | 51% [40%, 62%] | 0.03 [-0.19, 0.24] | 55% [44%, 65%] | 50% | 25% |
|
|
80
|
+
|
|
81
|
+
n = 80 pairs per judge; brackets are 95% intervals. No judge passed. The closest,
|
|
82
|
+
qwen2.5 7B, reaches kappa 0.40 to 0.45 with the human experts, but its verdict
|
|
83
|
+
flips on a quarter of the pairs when the answer order is swapped, and setting the
|
|
84
|
+
temperature to 0 removes the run-to-run flips without removing that order effect.
|
|
85
|
+
llama3.2 3B picked whichever answer it saw second 80% of the time. The longer-answer
|
|
86
|
+
rule never flips and agrees with humans 66% of the time, which is why stability and
|
|
87
|
+
accuracy alone are not enough: the padding probe catches it.
|
|
88
|
+
|
|
89
|
+
The same check on the suite's own pass/fail judge (qwen2.5 7B, 24 cases, no human
|
|
90
|
+
labels) gives `unstable`: its verdict changed across 5 identical calls on 7 of 24
|
|
91
|
+
cases (29%, interval 15% to 49%).
|
|
92
|
+
|
|
93
|
+
## How it works
|
|
94
|
+
|
|
95
|
+
| step | what it does | what it uses |
|
|
96
|
+
|---|---|---|
|
|
97
|
+
| `ingest` | Reads OpenAI chat JSONL (fine-tuning style or request/response logs), Anthropic messages, Langfuse trace exports, OpenTelemetry GenAI spans (OTLP JSON, current `gen_ai.input.messages` and older `gen_ai.prompt.N` attributes), or generic input/output JSONL. Normalizes to one schema with input, output, earlier turns, tools called, error flag, user feedback, route and model. Redacts emails, API keys, Luhn-valid card numbers, phone numbers and SSN-shaped numbers by default and reports counts. | Python standard library |
|
|
98
|
+
| `select` | Removes exact duplicates (normalized hash of the user turns) and near-duplicates (character n-gram cosine), clusters the rest by topic, then picks: failures first (errors, negative feedback), at least one case per stratum (route, tool, feedback, any metadata key you name), one central case per cluster, and fills the rest by cluster size with the least similar remaining cases. Every pick records why it was picked. | scikit-learn (TF-IDF, k-means) |
|
|
99
|
+
| `draft`, `validate` | Writes `cases.yaml` and `rubric.yaml` with TODO markers. The agent fills in expected behavior and criteria with you. Validation refuses ready cases that still contain TODO or reference unknown criteria. | PyYAML |
|
|
100
|
+
| `judge-plan` | Lists every judge call to make: each case N times, plus probes that swap the answer order (pairwise judges) and pad an answer with an irrelevant paragraph. | |
|
|
101
|
+
| `judge-run` | Optional and off by default. Sends each request as a JSON line to a command you name (your script, your provider, your keys) and records the verdicts. eval-builder ships no API keys and no provider code. | your command |
|
|
102
|
+
| `judge-check` | Per judge: flip rate across repeated calls (with a Wilson interval), self-agreement, majority-of-3 vote stability, accuracy and Cohen's kappa against your human labels (with intervals), position consistency and first-shown preference, and how often padding moved the verdict toward the padded answer. Verdict: `trustworthy`, `unstable`, `biased`, `misaligned`, or `not_enough_data`, with the numbers behind it. | |
|
|
103
|
+
| `export` | promptfoo `promptfooconfig.yaml` (llm-rubric asserts), DeepEval dataset plus a `deepeval test run` file, Inspect AI dataset plus `task.py`, plain JSONL. The manifest lists file hashes and which judges passed. | |
|
|
104
|
+
| `report` | `report.md` and `report.json`: sources with sha256, counts, redactions, selection reasons, the judge table, and the limits. | |
|
|
105
|
+
|
|
106
|
+
The verdict thresholds are explicit flags with defaults: flip rate at most 20% of cases,
|
|
107
|
+
position consistency at least 80%, padding helps at most 10% of cases, kappa at least
|
|
108
|
+
0.4 on at least 20 human-labeled cases. A judge without human labels is never called
|
|
109
|
+
trustworthy.
|
|
110
|
+
|
|
111
|
+
## Setup for agents
|
|
112
|
+
|
|
113
|
+
```sh
|
|
114
|
+
uvx eval-builder setup # shows what it would change
|
|
115
|
+
uvx eval-builder setup --yes # applies it
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
`setup` registers the MCP server (`uvx eval-builder mcp`) with Claude Code (`claude mcp
|
|
119
|
+
add --scope user`), Codex (`[mcp_servers.eval-builder]` in `~/.codex/config.toml`) and
|
|
120
|
+
Cursor (`~/.cursor/mcp.json`), and copies the agent instructions to
|
|
121
|
+
`~/.claude/skills/eval-builder/` and `~/.codex/skills/eval-builder/`. It backs up any
|
|
122
|
+
file it edits and does nothing on a second run. The workflow the agent follows is in
|
|
123
|
+
[`skills/eval-builder/SKILL.md`](skills/eval-builder/SKILL.md).
|
|
124
|
+
|
|
125
|
+
MCP tools: `ingest`, `select`, `draft`, `list_cases`, `update_case`, `set_rubric`,
|
|
126
|
+
`validate`, `judge_plan`, `judge_check`, `export`, `report`, `status`. The judge runner
|
|
127
|
+
is CLI only, because it executes a command.
|
|
128
|
+
|
|
129
|
+
## What it can't do
|
|
130
|
+
|
|
131
|
+
- It does not write expected behavior or human labels. The agent drafts expected
|
|
132
|
+
behavior with you; labels must come from people. Without labels, judge-check can
|
|
133
|
+
tell you a judge is unstable or biased, but not that it is right.
|
|
134
|
+
- Selection is lexical. Two requests that mean the same thing in different words can
|
|
135
|
+
land in different clusters, and near-duplicate detection only catches close textual
|
|
136
|
+
matches.
|
|
137
|
+
- Redaction is pattern-based. Names, street addresses and free-form secrets get
|
|
138
|
+
through. Look at `traces.jsonl` before sharing a workspace.
|
|
139
|
+
- The bias probes cover answer order and irrelevant length only. Self-preference,
|
|
140
|
+
style bias and rubric misreadings are not measured.
|
|
141
|
+
- Small samples give wide intervals. The report prints them; read them.
|
|
142
|
+
- It does not run your app or your eval. The exported files do that in promptfoo,
|
|
143
|
+
DeepEval or Inspect AI.
|
|
144
|
+
|
|
145
|
+
## Privacy and safety
|
|
146
|
+
|
|
147
|
+
Everything runs locally. eval-builder makes no network calls and sends nothing
|
|
148
|
+
anywhere. The only process it starts is the judge command you name with
|
|
149
|
+
`judge-run --enable-judge-plugin`. Redaction is on unless you pass `--no-redact`, and
|
|
150
|
+
the report lists what was replaced (counts and kinds, never the values).
|
|
151
|
+
|
|
152
|
+
## License
|
|
153
|
+
|
|
154
|
+
MIT. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
# eval-builder
|
|
2
|
+
|
|
3
|
+
Your agent turns your app's real logs into an eval suite, and tells you which of its judges you can actually trust.
|
|
4
|
+
|
|
5
|
+
```sh
|
|
6
|
+
uvx eval-builder ingest ./logs # OpenAI, Anthropic, Langfuse, OpenTelemetry or JSONL
|
|
7
|
+
uvx eval-builder select -n 40 && uvx eval-builder draft
|
|
8
|
+
uvx eval-builder judge-check # after your agent has run each judge a few times
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
> Not on PyPI yet. Until the first release, run it straight from GitHub by replacing `uvx eval-builder` with
|
|
12
|
+
> `uvx --from git+https://github.com/Abelo9996/eval-builder eval-builder`. `setup` registers `uvx eval-builder mcp`, so it works once the
|
|
13
|
+
> package is on PyPI.
|
|
14
|
+
|
|
15
|
+
eval-builder is not another eval platform. It builds the suite and checks the judges,
|
|
16
|
+
then exports to the tools you already run: promptfoo, DeepEval, Inspect AI, or plain
|
|
17
|
+
JSONL. It never calls a model. Your coding agent (Claude Code, Codex, Cursor) does the
|
|
18
|
+
thinking through the MCP server; eval-builder does the selection, the bookkeeping and
|
|
19
|
+
the statistics, and writes down the evidence.
|
|
20
|
+
|
|
21
|
+
## Example: real output
|
|
22
|
+
|
|
23
|
+
Everything below comes from runs on this machine (Apple M4, 16 GB) that are committed
|
|
24
|
+
in [`examples/mt-bench/`](examples/mt-bench/). The data is the LMSYS
|
|
25
|
+
[MT-Bench human judgments](https://huggingface.co/datasets/lmsys/mt_bench_human_judgments)
|
|
26
|
+
dataset (CC BY 4.0): 80 two-turn questions answered by 6 models, plus pairwise verdicts
|
|
27
|
+
from expert human judges.
|
|
28
|
+
|
|
29
|
+
**Logs to suite.** `ingest` read 480 conversations (OpenAI chat format, redaction on,
|
|
30
|
+
0 matches). `select -n 24 --stratify category` removed 400 exact duplicates (each
|
|
31
|
+
question was asked to six models), clustered the 80 unique conversations into 9 topics
|
|
32
|
+
and picked 24 covering all 8 categories, 19 of them failures. From
|
|
33
|
+
[`suite/report.md`](examples/mt-bench/suite/report.md):
|
|
34
|
+
|
|
35
|
+
```
|
|
36
|
+
| trace | cluster | represents | why it was picked |
|
|
37
|
+
| q81-alpaca-13b | 0 | 6 | failure (negative user feedback); 62% of unique traces are failures and at least 30% of picks are reserved for them |
|
|
38
|
+
| q82-alpaca-13b | 0 | 6 | adds variety within cluster 0 (11 traces, 14%; response, previous response, previous); least similar to cases already picked there |
|
|
39
|
+
| q156-alpaca-13b | 3 | 6 | covers category=humanities (10 unique traces, 12%) |
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
The agent wrote expected behavior for the 24 cases, and `export` produced files that
|
|
43
|
+
Inspect AI 0.3.277, DeepEval 4.2.8 and promptfoo 0.124.0 all load (see
|
|
44
|
+
[`verify_exports.sh`](examples/mt-bench/verify_exports.sh)).
|
|
45
|
+
|
|
46
|
+
**Which judges can you trust?** Five local LLM judge setups (four models through
|
|
47
|
+
Ollama, one of them also at temperature 0) and two controls that are not language models judged 80 answer pairs where human experts
|
|
48
|
+
picked a winner. Each judge ran 5 times per pair, 5 more with the answers swapped, and
|
|
49
|
+
5 more with an irrelevant paragraph appended to one answer: 8,400 calls, 0 errors.
|
|
50
|
+
`eval-builder judge-check`, from [`judges/report.md`](examples/mt-bench/judges/report.md):
|
|
51
|
+
|
|
52
|
+
| judge | verdict | flip rate | accuracy vs humans | kappa | survives answer swap | first-shown answer picked | padding helped |
|
|
53
|
+
|---|---|---|---|---|---|---|---|
|
|
54
|
+
| qwen2.5:7b-instruct, temp 0 | biased | 0% [0%, 5%] | 72% [62%, 81%] | 0.45 [0.26, 0.65] | 75% [65%, 83%] | 44% | 0% |
|
|
55
|
+
| qwen2.5:7b-instruct, temp 0.8 | biased | 12% [7%, 22%] | 70% [59%, 79%] | 0.40 [0.20, 0.60] | 75% [65%, 83%] | 42% | 1% |
|
|
56
|
+
| qwen2.5:3b-instruct | biased | 12% [7%, 22%] | 56% [45%, 67%] | 0.13 [-0.08, 0.35] | 44% [33%, 55%] | 35% | 9% |
|
|
57
|
+
| gemma2:2b | unstable | 59% [48%, 69%] | 56% [45%, 67%] | 0.14 [-0.07, 0.36] | 38% [28%, 48%] | 35% | 11% |
|
|
58
|
+
| llama3.2:3b | unstable | 56% [45%, 67%] | 51% [40%, 62%] | 0.04 [-0.17, 0.26] | 14% [8%, 23%] | 20% | 10% |
|
|
59
|
+
| control: longer answer wins | biased | 0% [0%, 5%] | 66% [55%, 76%] | 0.32 [0.12, 0.53] | 100% [95%, 100%] | 50% | 30% |
|
|
60
|
+
| control: coin flip | unstable | 95% [88%, 98%] | 51% [40%, 62%] | 0.03 [-0.19, 0.24] | 55% [44%, 65%] | 50% | 25% |
|
|
61
|
+
|
|
62
|
+
n = 80 pairs per judge; brackets are 95% intervals. No judge passed. The closest,
|
|
63
|
+
qwen2.5 7B, reaches kappa 0.40 to 0.45 with the human experts, but its verdict
|
|
64
|
+
flips on a quarter of the pairs when the answer order is swapped, and setting the
|
|
65
|
+
temperature to 0 removes the run-to-run flips without removing that order effect.
|
|
66
|
+
llama3.2 3B picked whichever answer it saw second 80% of the time. The longer-answer
|
|
67
|
+
rule never flips and agrees with humans 66% of the time, which is why stability and
|
|
68
|
+
accuracy alone are not enough: the padding probe catches it.
|
|
69
|
+
|
|
70
|
+
The same check on the suite's own pass/fail judge (qwen2.5 7B, 24 cases, no human
|
|
71
|
+
labels) gives `unstable`: its verdict changed across 5 identical calls on 7 of 24
|
|
72
|
+
cases (29%, interval 15% to 49%).
|
|
73
|
+
|
|
74
|
+
## How it works
|
|
75
|
+
|
|
76
|
+
| step | what it does | what it uses |
|
|
77
|
+
|---|---|---|
|
|
78
|
+
| `ingest` | Reads OpenAI chat JSONL (fine-tuning style or request/response logs), Anthropic messages, Langfuse trace exports, OpenTelemetry GenAI spans (OTLP JSON, current `gen_ai.input.messages` and older `gen_ai.prompt.N` attributes), or generic input/output JSONL. Normalizes to one schema with input, output, earlier turns, tools called, error flag, user feedback, route and model. Redacts emails, API keys, Luhn-valid card numbers, phone numbers and SSN-shaped numbers by default and reports counts. | Python standard library |
|
|
79
|
+
| `select` | Removes exact duplicates (normalized hash of the user turns) and near-duplicates (character n-gram cosine), clusters the rest by topic, then picks: failures first (errors, negative feedback), at least one case per stratum (route, tool, feedback, any metadata key you name), one central case per cluster, and fills the rest by cluster size with the least similar remaining cases. Every pick records why it was picked. | scikit-learn (TF-IDF, k-means) |
|
|
80
|
+
| `draft`, `validate` | Writes `cases.yaml` and `rubric.yaml` with TODO markers. The agent fills in expected behavior and criteria with you. Validation refuses ready cases that still contain TODO or reference unknown criteria. | PyYAML |
|
|
81
|
+
| `judge-plan` | Lists every judge call to make: each case N times, plus probes that swap the answer order (pairwise judges) and pad an answer with an irrelevant paragraph. | |
|
|
82
|
+
| `judge-run` | Optional and off by default. Sends each request as a JSON line to a command you name (your script, your provider, your keys) and records the verdicts. eval-builder ships no API keys and no provider code. | your command |
|
|
83
|
+
| `judge-check` | Per judge: flip rate across repeated calls (with a Wilson interval), self-agreement, majority-of-3 vote stability, accuracy and Cohen's kappa against your human labels (with intervals), position consistency and first-shown preference, and how often padding moved the verdict toward the padded answer. Verdict: `trustworthy`, `unstable`, `biased`, `misaligned`, or `not_enough_data`, with the numbers behind it. | |
|
|
84
|
+
| `export` | promptfoo `promptfooconfig.yaml` (llm-rubric asserts), DeepEval dataset plus a `deepeval test run` file, Inspect AI dataset plus `task.py`, plain JSONL. The manifest lists file hashes and which judges passed. | |
|
|
85
|
+
| `report` | `report.md` and `report.json`: sources with sha256, counts, redactions, selection reasons, the judge table, and the limits. | |
|
|
86
|
+
|
|
87
|
+
The verdict thresholds are explicit flags with defaults: flip rate at most 20% of cases,
|
|
88
|
+
position consistency at least 80%, padding helps at most 10% of cases, kappa at least
|
|
89
|
+
0.4 on at least 20 human-labeled cases. A judge without human labels is never called
|
|
90
|
+
trustworthy.
|
|
91
|
+
|
|
92
|
+
## Setup for agents
|
|
93
|
+
|
|
94
|
+
```sh
|
|
95
|
+
uvx eval-builder setup # shows what it would change
|
|
96
|
+
uvx eval-builder setup --yes # applies it
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
`setup` registers the MCP server (`uvx eval-builder mcp`) with Claude Code (`claude mcp
|
|
100
|
+
add --scope user`), Codex (`[mcp_servers.eval-builder]` in `~/.codex/config.toml`) and
|
|
101
|
+
Cursor (`~/.cursor/mcp.json`), and copies the agent instructions to
|
|
102
|
+
`~/.claude/skills/eval-builder/` and `~/.codex/skills/eval-builder/`. It backs up any
|
|
103
|
+
file it edits and does nothing on a second run. The workflow the agent follows is in
|
|
104
|
+
[`skills/eval-builder/SKILL.md`](skills/eval-builder/SKILL.md).
|
|
105
|
+
|
|
106
|
+
MCP tools: `ingest`, `select`, `draft`, `list_cases`, `update_case`, `set_rubric`,
|
|
107
|
+
`validate`, `judge_plan`, `judge_check`, `export`, `report`, `status`. The judge runner
|
|
108
|
+
is CLI only, because it executes a command.
|
|
109
|
+
|
|
110
|
+
## What it can't do
|
|
111
|
+
|
|
112
|
+
- It does not write expected behavior or human labels. The agent drafts expected
|
|
113
|
+
behavior with you; labels must come from people. Without labels, judge-check can
|
|
114
|
+
tell you a judge is unstable or biased, but not that it is right.
|
|
115
|
+
- Selection is lexical. Two requests that mean the same thing in different words can
|
|
116
|
+
land in different clusters, and near-duplicate detection only catches close textual
|
|
117
|
+
matches.
|
|
118
|
+
- Redaction is pattern-based. Names, street addresses and free-form secrets get
|
|
119
|
+
through. Look at `traces.jsonl` before sharing a workspace.
|
|
120
|
+
- The bias probes cover answer order and irrelevant length only. Self-preference,
|
|
121
|
+
style bias and rubric misreadings are not measured.
|
|
122
|
+
- Small samples give wide intervals. The report prints them; read them.
|
|
123
|
+
- It does not run your app or your eval. The exported files do that in promptfoo,
|
|
124
|
+
DeepEval or Inspect AI.
|
|
125
|
+
|
|
126
|
+
## Privacy and safety
|
|
127
|
+
|
|
128
|
+
Everything runs locally. eval-builder makes no network calls and sends nothing
|
|
129
|
+
anywhere. The only process it starts is the judge command you name with
|
|
130
|
+
`judge-run --enable-judge-plugin`. Redaction is on unless you pass `--no-redact`, and
|
|
131
|
+
the report lists what was replaced (counts and kinds, never the values).
|
|
132
|
+
|
|
133
|
+
## License
|
|
134
|
+
|
|
135
|
+
MIT. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Security
|
|
2
|
+
|
|
3
|
+
eval-builder reads log files you point it at and writes files into a workspace
|
|
4
|
+
directory. It makes no network calls. The optional judge runner starts only the
|
|
5
|
+
command you pass with `--command`, and only when you also pass
|
|
6
|
+
`--enable-judge-plugin`.
|
|
7
|
+
|
|
8
|
+
Logs often contain personal data. Redaction is on by default but is pattern-based and
|
|
9
|
+
will miss things (names, addresses, free-form secrets). Review the workspace before
|
|
10
|
+
sharing it.
|
|
11
|
+
|
|
12
|
+
To report a vulnerability, open a private security advisory on the GitHub repository
|
|
13
|
+
or email the maintainer. Please do not file a public issue for security problems.
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""Control judges for `eval-builder judge-run`. These are NOT language models.
|
|
2
|
+
|
|
3
|
+
--kind longer picks whichever answer is longer (a deterministic heuristic)
|
|
4
|
+
--kind coin picks A or B at random (seeded by request id), a pure-noise control
|
|
5
|
+
|
|
6
|
+
They exist to show that judge-check separates "stable but biased" (longer) and
|
|
7
|
+
"noise" (coin) from real judges. Never report them as LLM judges.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import argparse
|
|
13
|
+
import hashlib
|
|
14
|
+
import json
|
|
15
|
+
import sys
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def main() -> None:
|
|
19
|
+
ap = argparse.ArgumentParser()
|
|
20
|
+
ap.add_argument("--kind", choices=["longer", "coin"], required=True)
|
|
21
|
+
args = ap.parse_args()
|
|
22
|
+
for line in sys.stdin:
|
|
23
|
+
req = json.loads(line)
|
|
24
|
+
shown = req["presented"]
|
|
25
|
+
if args.kind == "longer":
|
|
26
|
+
verdict = "A" if len(shown["answer_a"]) >= len(shown["answer_b"]) else "B"
|
|
27
|
+
else:
|
|
28
|
+
h = hashlib.sha256(req["request_id"].encode()).digest()[0]
|
|
29
|
+
verdict = "A" if h % 2 == 0 else "B"
|
|
30
|
+
print(json.dumps({"verdict": verdict}), flush=True)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
if __name__ == "__main__":
|
|
34
|
+
main()
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""Example judge plugin for `eval-builder judge-run`: a local model served by Ollama.
|
|
2
|
+
|
|
3
|
+
This is an example, not part of the eval-builder package. It talks only to the
|
|
4
|
+
Ollama server on localhost. Protocol: read one JSON request per line on stdin,
|
|
5
|
+
write one JSON response per line on stdout: {"verdict": ..., "raw": ...}.
|
|
6
|
+
|
|
7
|
+
Usage:
|
|
8
|
+
eval-builder judge-run --enable-judge-plugin \\
|
|
9
|
+
--command "qwen-3b=python examples/judges/ollama_judge.py --model qwen2.5:3b-instruct"
|
|
10
|
+
|
|
11
|
+
Each trial uses seed = trial index, so a rerun on the same machine, model digest and
|
|
12
|
+
Ollama version reproduces the same outputs.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import argparse
|
|
18
|
+
import json
|
|
19
|
+
import sys
|
|
20
|
+
import urllib.request
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def generate(
|
|
24
|
+
host: str,
|
|
25
|
+
model: str,
|
|
26
|
+
prompt: str,
|
|
27
|
+
temperature: float,
|
|
28
|
+
seed: int,
|
|
29
|
+
num_ctx: int,
|
|
30
|
+
num_predict: int,
|
|
31
|
+
) -> dict:
|
|
32
|
+
body = json.dumps(
|
|
33
|
+
{
|
|
34
|
+
"model": model,
|
|
35
|
+
"prompt": prompt,
|
|
36
|
+
"stream": False,
|
|
37
|
+
"options": {
|
|
38
|
+
"temperature": temperature,
|
|
39
|
+
"seed": seed,
|
|
40
|
+
"num_ctx": num_ctx,
|
|
41
|
+
"num_predict": num_predict,
|
|
42
|
+
},
|
|
43
|
+
}
|
|
44
|
+
).encode()
|
|
45
|
+
req = urllib.request.Request(f"{host}/api/generate", body, {"Content-Type": "application/json"})
|
|
46
|
+
with urllib.request.urlopen(req, timeout=300) as resp:
|
|
47
|
+
return json.load(resp)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def main() -> None:
|
|
51
|
+
ap = argparse.ArgumentParser()
|
|
52
|
+
ap.add_argument("--model", required=True)
|
|
53
|
+
ap.add_argument("--temperature", type=float, default=0.8)
|
|
54
|
+
ap.add_argument("--num-ctx", type=int, default=8192)
|
|
55
|
+
ap.add_argument("--num-predict", type=int, default=8)
|
|
56
|
+
ap.add_argument("--host", default="http://127.0.0.1:11434")
|
|
57
|
+
args = ap.parse_args()
|
|
58
|
+
for line in sys.stdin:
|
|
59
|
+
req = json.loads(line)
|
|
60
|
+
if not req.get("prompt"):
|
|
61
|
+
print(
|
|
62
|
+
json.dumps({"verdict": None, "raw": "request has no rendered prompt"}), flush=True
|
|
63
|
+
)
|
|
64
|
+
continue
|
|
65
|
+
out = generate(
|
|
66
|
+
args.host,
|
|
67
|
+
args.model,
|
|
68
|
+
req["prompt"],
|
|
69
|
+
args.temperature,
|
|
70
|
+
int(req.get("trial", 0)),
|
|
71
|
+
args.num_ctx,
|
|
72
|
+
args.num_predict,
|
|
73
|
+
)
|
|
74
|
+
text = out.get("response", "")
|
|
75
|
+
print(
|
|
76
|
+
json.dumps(
|
|
77
|
+
{
|
|
78
|
+
"verdict": text.strip(),
|
|
79
|
+
"raw": text,
|
|
80
|
+
"prompt_tokens": out.get("prompt_eval_count"),
|
|
81
|
+
}
|
|
82
|
+
),
|
|
83
|
+
flush=True,
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
if __name__ == "__main__":
|
|
88
|
+
main()
|