eval-builder 0.1.1__tar.gz → 0.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {eval_builder-0.1.1 → eval_builder-0.1.2}/.gitignore +2 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/AGENTS.md +5 -1
- {eval_builder-0.1.1 → eval_builder-0.1.2}/CHANGELOG.md +40 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/PKG-INFO +76 -13
- {eval_builder-0.1.1 → eval_builder-0.1.2}/README.md +75 -12
- eval_builder-0.1.2/docs/demo.gif +0 -0
- eval_builder-0.1.2/docs/demo.tape +25 -0
- eval_builder-0.1.2/docs/label-sheet.png +0 -0
- eval_builder-0.1.2/examples/sample-labeling/README.md +79 -0
- eval_builder-0.1.2/examples/sample-labeling/decisions.json +98 -0
- eval_builder-0.1.2/examples/sample-labeling/drive_sheet.py +90 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/cases.yaml +2042 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/exports/deepeval/dataset.json +1145 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/exports/deepeval/judge.json +10 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/exports/deepeval/test_eval_builder.py +165 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/exports/inspect/dataset.jsonl +47 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/exports/inspect/task.py +23 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/exports/jsonl/cases.jsonl +47 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/exports/manifest.json +90 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/exports/promptfoo/promptfooconfig.yaml +2314 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/ingest.json +33 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/judge_check.json +2188 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/judge_run_log.json +35 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/judgments.jsonl +1128 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/label_plan.json +673 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/label_sheet.csv +313 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/label_sheet.html +445 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/labels.jsonl +24 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/report.json +3593 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/report.md +148 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/rubric.yaml +100 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/selection.json +1227 -0
- eval_builder-0.1.2/examples/sample-labeling/evalset/traces.jsonl +48 -0
- eval_builder-0.1.2/examples/sample-labeling/fill.py +127 -0
- eval_builder-0.1.2/examples/sample-labeling/run.sh +35 -0
- eval_builder-0.1.2/examples/sample-labeling/sheet-run/drive_log.txt +12 -0
- eval_builder-0.1.2/examples/sample-labeling/sheet-run/labels.jsonl +24 -0
- eval_builder-0.1.2/examples/sample-labeling/sheet-run/sheet-export.png +0 -0
- eval_builder-0.1.2/examples/sample-labeling/sheet-run/sheet-first-case.png +0 -0
- eval_builder-0.1.2/examples/sample-labeling/sheet-run/sheet-phone.png +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/pyproject.toml +5 -1
- {eval_builder-0.1.1 → eval_builder-0.1.2}/skills/eval-builder/SKILL.md +28 -13
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/__init__.py +1 -1
- eval_builder-0.1.2/src/eval_builder/balance.py +55 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/cli.py +88 -2
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/export.py +199 -22
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/judge/check.py +69 -10
- eval_builder-0.1.2/src/eval_builder/label.py +549 -0
- eval_builder-0.1.2/src/eval_builder/label_sheet.py +481 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/mcp_server.py +43 -9
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/report.py +14 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/select.py +24 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/status.py +9 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/workspace.py +12 -0
- eval_builder-0.1.2/tests/test_label.py +349 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/uv.lock +1 -1
- {eval_builder-0.1.1 → eval_builder-0.1.2}/.github/workflows/ci.yml +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/.github/workflows/release.yml +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/CONTRIBUTING.md +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/LICENSE +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/SECURITY.md +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/judges/control_judges.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/judges/ollama_judge.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/README.md +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/expected_behaviors.yaml +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/fill_suite.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/judges/cases.yaml +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/judges/judge_check.json +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/judges/judge_run_log.json +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/judges/judgments.jsonl +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/judges/labels.jsonl +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/judges/report.json +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/judges/report.md +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/judges/rubric.yaml +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/prepare.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/run.sh +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/cases.yaml +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/exports/deepeval/dataset.json +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/exports/deepeval/test_eval_builder.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/exports/inspect/dataset.jsonl +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/exports/inspect/task.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/exports/jsonl/cases.jsonl +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/exports/manifest.json +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/exports/promptfoo/promptfooconfig.yaml +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/ingest.json +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/judge_check.json +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/judge_run_log.json +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/judgments.jsonl +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/report.json +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/report.md +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/rubric.yaml +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/selection.json +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/verify_exports.sh +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/sample_logs.jsonl +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/draft.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/ingest/__init__.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/ingest/formats.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/io.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/judge/__init__.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/judge/plan.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/judge/run.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/judge/stats.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/redact.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/schema.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/setup_agents.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/conftest.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/fixtures/anthropic_messages.json +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/fixtures/generic.jsonl +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/fixtures/langfuse_export.json +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/fixtures/openai_chat.jsonl +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/fixtures/otel_genai.json +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_cli_mcp.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_draft.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_export.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_ingest.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_judge_check.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_judge_run.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_judge_stats.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_report.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_select.py +0 -0
- {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_setup.py +0 -0
|
@@ -9,9 +9,11 @@ build/
|
|
|
9
9
|
*.egg-info/
|
|
10
10
|
.DS_Store
|
|
11
11
|
evalset/
|
|
12
|
+
!examples/sample-labeling/evalset/
|
|
12
13
|
.deepeval/
|
|
13
14
|
# Example inputs that prepare.py downloads or derives (reproducible from the pinned dataset)
|
|
14
15
|
examples/mt-bench/data/
|
|
15
16
|
# Large or derived files inside example workspaces
|
|
16
17
|
examples/mt-bench/*/judge_requests.jsonl
|
|
17
18
|
examples/mt-bench/*/traces.jsonl
|
|
19
|
+
examples/sample-labeling/evalset/judge_requests.jsonl
|
|
@@ -16,6 +16,9 @@ src/eval_builder/
|
|
|
16
16
|
judge/run.py opt-in judge plugin runner (off by default, runs a user command)
|
|
17
17
|
judge/check.py flip rate, kappa, accuracy, probes, verdicts
|
|
18
18
|
judge/stats.py Wilson interval, Cohen's kappa, majority vote
|
|
19
|
+
label.py which cases a person should label, the CSV sheet, `label import`
|
|
20
|
+
label_sheet.py the offline HTML labeling sheet (one file, inline JS, no network)
|
|
21
|
+
balance.py the warning when cases or labels are mostly one outcome
|
|
19
22
|
export.py promptfoo, DeepEval, Inspect AI, JSONL
|
|
20
23
|
report.py report.md and report.json
|
|
21
24
|
setup_agents.py `eval-builder setup` for Claude Code, Codex, Cursor
|
|
@@ -25,7 +28,8 @@ src/eval_builder/
|
|
|
25
28
|
|
|
26
29
|
## Rules
|
|
27
30
|
|
|
28
|
-
- No model calls and no network access in the package.
|
|
31
|
+
- No model calls and no network access in the package. (Exported files may call a
|
|
32
|
+
model when the user runs them, for example the DeepEval test calling a wired judge.) The judge runner only starts a
|
|
29
33
|
command the user names, and only with an explicit flag.
|
|
30
34
|
- Deterministic: same inputs and seed give the same selection and the same files.
|
|
31
35
|
- Every number the tool reports must be traceable to a file in the workspace.
|
|
@@ -1,5 +1,45 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.1.2 (2026-10-08)
|
|
4
|
+
|
|
5
|
+
Human labels. No judge can be called trustworthy without them, and both real agent
|
|
6
|
+
sessions against 0.1.0 and 0.1.1 stopped at that step: they asked for labels and had
|
|
7
|
+
no way to collect them.
|
|
8
|
+
|
|
9
|
+
- `label` (CLI, and MCP `label`): picks the ready cases a person should label, 24 by
|
|
10
|
+
default (judge-check needs 20). The budget is split evenly across outcomes (the
|
|
11
|
+
judges' consensus verdict, or the logged failure flag before any judge has run), and
|
|
12
|
+
up to half of each share goes to cases where the judges disagree, flip across
|
|
13
|
+
repeats or move under padding. Every pick and its reason is in `label_plan.json`.
|
|
14
|
+
- `label` writes `label_sheet.html`: one self-contained file that works offline and
|
|
15
|
+
makes no network requests (its content security policy blocks them). One case per
|
|
16
|
+
screen with the earlier turns, the reply, the expected behavior and criteria;
|
|
17
|
+
pass/fail or A/B buttons (labels come from rubric.yaml), an optional note, keyboard
|
|
18
|
+
shortcuts, progress, and the judges' verdicts hidden. Progress survives a reload
|
|
19
|
+
through the browser's local storage. Export downloads `labels.jsonl` in exactly the
|
|
20
|
+
format judge-check reads and shows the same text to copy.
|
|
21
|
+
- `label` also writes `label_sheet.csv` for spreadsheet users (fill the label column).
|
|
22
|
+
- `label import <file>` (CLI, and MCP `label_import`): reads the sheet's labels.jsonl,
|
|
23
|
+
the filled-in CSV, a JSON list, or stdin (`-`); checks every case id and label
|
|
24
|
+
against cases.yaml and rubric.yaml; merges into `labels.jsonl` (a new label replaces
|
|
25
|
+
the same labeler's earlier one, other labelers are kept); lists rejected rows with
|
|
26
|
+
the reason and exits 1 when there are any.
|
|
27
|
+
- `select` and `judge-check` warn, with the real proportions, when at least 80% of the
|
|
28
|
+
selected cases or of the human labels share one outcome. judge-check also warns when
|
|
29
|
+
a judge gave the same verdict on every labeled case or its accuracy is no better
|
|
30
|
+
than always giving the most common label, and reports `majority_baseline` (that
|
|
31
|
+
always-the-common-label accuracy).
|
|
32
|
+
- DeepEval export: a pointwise judge that passed judge-check (or one forced with
|
|
33
|
+
`--judge`) now grades every case with its exact rubric.yaml prompt, through a custom
|
|
34
|
+
metric in `test_eval_builder.py` and `judge.json`. Ollama providers are called on the
|
|
35
|
+
local server; for other providers you fill in `call_judge`. Without a checked judge
|
|
36
|
+
the file keeps GEval. The export notes and README say that Inspect AI still uses its
|
|
37
|
+
default `model_graded_qa` grader.
|
|
38
|
+
- judge-check parses JSON verdicts that a token limit cut off mid-reason, and verdicts
|
|
39
|
+
wrapped in a json code fence.
|
|
40
|
+
- `status` and judge-check's `next` walk through the labeling step.
|
|
41
|
+
- `examples/sample-labeling/`: the whole flow on the bundled 48-conversation sample.
|
|
42
|
+
|
|
3
43
|
## 0.1.1 (2026-10-08)
|
|
4
44
|
|
|
5
45
|
Fixes from a fresh-install audit and a real Claude Code session driving the MCP server.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: eval-builder
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.2
|
|
4
4
|
Summary: Turn real LLM app logs into an eval suite and measure which LLM judges you can trust.
|
|
5
5
|
Author: Abel Yagubyan
|
|
6
6
|
License-Expression: MIT
|
|
@@ -29,18 +29,26 @@ uvx eval-builder ingest sample_logs.jsonl # or your own logs: OpenAI, Anthro
|
|
|
29
29
|
uvx eval-builder select -n 10 --stratify category && uvx eval-builder draft
|
|
30
30
|
```
|
|
31
31
|
|
|
32
|
-
|
|
32
|
+

|
|
33
|
+
|
|
34
|
+
What you get (real output of the commands above, eval-builder 0.1.1 from PyPI; the
|
|
35
|
+
curl and the three uvx calls took 35 s in total here; the very first uvx run also
|
|
36
|
+
downloads about 38 MiB, mostly numpy, scipy and scikit-learn, which took 6 s here):
|
|
33
37
|
|
|
34
38
|
```
|
|
35
39
|
ingested 48 traces into evalset/traces.jsonl
|
|
36
40
|
sample_logs.jsonl: format=openai records=48 traces=48 skipped=0 sha256=2d5dd2295784988b
|
|
37
41
|
redactions: 0 {}
|
|
42
|
+
next: eval-builder select -n 30 (add --stratify <metadata keys> to cover them)
|
|
38
43
|
48 traces -> 16 unique (32 exact dupes, 0 near dupes) -> selected 10 (8 failures) across 4 clusters
|
|
39
44
|
q121-llama-13b: failure (negative user feedback); 69% of unique traces are failures and at least 30% of picks are reserved for them
|
|
45
|
+
q81-alpaca-13b: failure (negative user feedback); 69% of unique traces are failures and at least 30% of picks are reserved for them
|
|
40
46
|
q102-alpaca-13b: covers category=reasoning (2 unique traces, 12%)
|
|
41
47
|
q111-alpaca-13b: adds variety within cluster 1 (8 traces, 50%; triangle, response, person); least similar to cases already picked there
|
|
42
|
-
...
|
|
48
|
+
... (6 more picks)
|
|
49
|
+
next: run draft to turn the selection into cases.yaml
|
|
43
50
|
10 case(s) added, 10 total in evalset/cases.yaml
|
|
51
|
+
next: read the cases (list_cases, or cases.yaml), define criteria and judges (set_rubric, or rubric.yaml), fill expected_behavior and criteria per case and set status: ready (update_case), then run validate
|
|
44
52
|
```
|
|
45
53
|
|
|
46
54
|
`evalset/cases.yaml` now holds 10 real conversations with a TODO where the expected
|
|
@@ -90,7 +98,10 @@ export and report in 32 turns and about 10 minutes. From its final answer:
|
|
|
90
98
|
> Its verdicts will be unverified.
|
|
91
99
|
|
|
92
100
|
It wrote every expected behavior itself because nobody was there to confirm them, and
|
|
93
|
-
said so on each case. It did not invent human labels; it asked for them.
|
|
101
|
+
said so on each case. It did not invent human labels; it asked for them. That is where
|
|
102
|
+
this session and the earlier one both stopped, so 0.1.2 adds the missing step: `label`
|
|
103
|
+
writes a sheet the person labels in their browser, and `label import` brings the
|
|
104
|
+
labels back for judge-check ([Human labels, end to end](#human-labels-end-to-end)).
|
|
94
105
|
|
|
95
106
|
An earlier session against 0.1.0 is where the promptfoo context bug fixed in 0.1.1 came
|
|
96
107
|
from. The agent read the export and told the user: "Right now it only sends the final
|
|
@@ -150,6 +161,42 @@ The same check on the suite's own pass/fail judge (qwen2.5 7B, 24 cases, no huma
|
|
|
150
161
|
labels) gives `unstable`: its verdict changed across 5 identical calls on 7 of 24
|
|
151
162
|
cases (29%, interval 15% to 49%).
|
|
152
163
|
|
|
164
|
+
### Human labels, end to end
|
|
165
|
+
|
|
166
|
+
```sh
|
|
167
|
+
uvx eval-builder label # picks 24 cases, writes label_sheet.html
|
|
168
|
+
uvx eval-builder label import ~/Downloads/labels.jsonl # what the sheet's Export button saved
|
|
169
|
+
uvx eval-builder judge-check
|
|
170
|
+
```
|
|
171
|
+
|
|
172
|
+

|
|
173
|
+
|
|
174
|
+
Run on the 48 sample conversations with every model's answer kept as a case (47
|
|
175
|
+
cases), three local judges (1,128 calls, 0 errors), and 24 labels entered through the
|
|
176
|
+
sheet in headless Chrome. The labels were made by the developer (Claude Code reading
|
|
177
|
+
each case for him), not by an independent annotator, so read this as a demonstration
|
|
178
|
+
of the flow. Details and every file: [`examples/sample-labeling/`](examples/sample-labeling/).
|
|
179
|
+
|
|
180
|
+
- `label` split the 24 picks 12/12 between cases the judges called pass and fail; 15
|
|
181
|
+
are cases where the judges disagree, flip or move under padding.
|
|
182
|
+
- The sheet made no network requests, survived a reload mid-way, and its download was
|
|
183
|
+
byte-identical to the copy box. `label import` took 24 rows, rejected 0.
|
|
184
|
+
- `judge-check` against those labels (fail 18, pass 6):
|
|
185
|
+
|
|
186
|
+
| judge | verdict | flip rate | accuracy vs labels | kappa |
|
|
187
|
+
|---|---|---|---|---|
|
|
188
|
+
| qwen2.5:7b-instruct, temp 0 | trustworthy | 0% [0%, 8%] | 75% [55%, 88%] | 0.50 [0.15, 0.85] |
|
|
189
|
+
| qwen2.5:7b-instruct, temp 0.8 | trustworthy | 9% [3%, 20%] | 71% [51%, 85%] | 0.44 [0.09, 0.79] |
|
|
190
|
+
| llama3.2:3b | unstable | 47% [33%, 61%] | 54% [35%, 72%] | 0.12 [-0.26, 0.50] |
|
|
191
|
+
|
|
192
|
+
The qwen judges pass the default thresholds, but judge-check also warns that their
|
|
193
|
+
accuracy is no better than always answering "fail" (75% of the labels), and every
|
|
194
|
+
miss went the same way: both passed a reply that said "Here is an allegorical poem"
|
|
195
|
+
and then wrote no poem. With 24 labels the kappa interval runs from about 0.1 to 0.8.
|
|
196
|
+
The DeepEval export wired the 0.8-temperature qwen judge in with its exact prompt;
|
|
197
|
+
DeepEval 4.2.8 ran three logged cases through it and it made the same call on the
|
|
198
|
+
missing poem.
|
|
199
|
+
|
|
153
200
|
## How it works
|
|
154
201
|
|
|
155
202
|
| step | what it does | what it uses |
|
|
@@ -159,14 +206,17 @@ cases (29%, interval 15% to 49%).
|
|
|
159
206
|
| `draft`, `validate` | Writes `cases.yaml` and `rubric.yaml` with TODO markers. The agent fills in expected behavior and criteria with you. Validation refuses ready cases that still contain TODO or reference unknown criteria. | PyYAML |
|
|
160
207
|
| `judge-plan` | Lists every judge call to make: each case N times, plus probes that swap the answer order (pairwise judges) and pad an answer with an irrelevant paragraph. | |
|
|
161
208
|
| `judge-run` | Optional and off by default. Sends each request as a JSON line to a command you name (your script, your provider, your keys) and records the verdicts. eval-builder ships no API keys and no provider code. | your command |
|
|
209
|
+
| `label`, `label import` | Picks the ready cases a person should label (default 24): the budget is split across outcomes (the judges' consensus, or the logged failure flag before judges ran) and up to half of each share goes to cases where judges disagree, flip across repeats or move under padding. Writes `label_sheet.html`, one offline file (one case per screen, pass/fail or A/B buttons, a note, keyboard shortcuts, judge verdicts hidden, progress kept in the browser) whose Export button downloads `labels.jsonl` in the format judge-check reads, and `label_sheet.csv` for spreadsheet users. `label import` checks the file against `cases.yaml` and merges it into the workspace. | Python standard library; the sheet is plain HTML and JavaScript |
|
|
162
210
|
| `judge-check` | Per judge: flip rate across repeated calls (with a Wilson interval), self-agreement, majority-of-3 vote stability, accuracy and Cohen's kappa against your human labels (with intervals), position consistency and first-shown preference, and how often padding moved the verdict toward the padded answer. Verdict: `trustworthy`, `unstable`, `biased`, `misaligned`, or `not_enough_data`, with the numbers behind it. | |
|
|
163
|
-
| `export` | promptfoo `promptfooconfig.yaml` (the full conversation as chat messages, llm-rubric asserts, and a judge that passed judge-check wired in as the grader when its rubric entry names a promptfoo `provider`), DeepEval dataset plus a `deepeval test run` file, Inspect AI dataset plus `task.py
|
|
211
|
+
| `export` | promptfoo `promptfooconfig.yaml` (the full conversation as chat messages, llm-rubric asserts, and a judge that passed judge-check wired in as the grader when its rubric entry names a promptfoo `provider`), DeepEval dataset plus a `deepeval test run` file (the same checked judge grades every case with its exact prompt; without one, GEval), Inspect AI dataset plus `task.py` (Inspect's default `model_graded_qa` grader), plain JSONL. The manifest lists file hashes and which judges passed. | |
|
|
164
212
|
| `report` | `report.md` and `report.json`: sources with sha256, counts, redactions, selection reasons, the judge table, and the limits. | |
|
|
165
213
|
|
|
166
214
|
The verdict thresholds are explicit flags with defaults: flip rate at most 20% of cases,
|
|
167
215
|
position consistency at least 80%, padding helps at most 10% of cases, kappa at least
|
|
168
216
|
0.4 on at least 20 human-labeled cases. A judge without human labels is never called
|
|
169
|
-
trustworthy.
|
|
217
|
+
trustworthy. When at least 80% of the selected cases or of the human labels share one
|
|
218
|
+
outcome, `select` and `judge-check` say so with the real proportions, because a judge
|
|
219
|
+
that always gives that answer would look accurate on them.
|
|
170
220
|
|
|
171
221
|
## Setup for agents
|
|
172
222
|
|
|
@@ -183,14 +233,22 @@ file it edits and does nothing on a second run. The workflow the agent follows i
|
|
|
183
233
|
[`skills/eval-builder/SKILL.md`](skills/eval-builder/SKILL.md).
|
|
184
234
|
|
|
185
235
|
MCP tools: `ingest`, `select`, `draft`, `list_cases`, `update_case`, `set_rubric`,
|
|
186
|
-
`validate`, `judge_plan`, `judge_check`, `export`, `report`,
|
|
236
|
+
`validate`, `judge_plan`, `label`, `label_import`, `judge_check`, `export`, `report`,
|
|
237
|
+
`status`. The judge runner
|
|
187
238
|
is CLI only, because it executes a command.
|
|
188
239
|
|
|
189
240
|
## What it can't do
|
|
190
241
|
|
|
191
242
|
- It does not write expected behavior or human labels. The agent drafts expected
|
|
192
243
|
behavior with you; labels must come from people. Without labels, judge-check can
|
|
193
|
-
tell you a judge is unstable or biased, but not that it is right.
|
|
244
|
+
tell you a judge is unstable or biased, but not that it is right. `label` makes the
|
|
245
|
+
labeling quick, but someone still has to read each case.
|
|
246
|
+
- Picking labels where judges disagree makes each label more informative, but those
|
|
247
|
+
cases are harder than average, so accuracy measured on them leans pessimistic.
|
|
248
|
+
`--uncertain-share 0` picks by outcome and topic only.
|
|
249
|
+
- The labeling sheet is a static page, so it cannot save files: the person has to
|
|
250
|
+
click Export (or copy the text) and import it. Unexported progress lives only in
|
|
251
|
+
that browser's local storage.
|
|
194
252
|
- Selection is lexical. Two requests that mean the same thing in different words can
|
|
195
253
|
land in different clusters, and near-duplicate detection only catches close textual
|
|
196
254
|
matches.
|
|
@@ -202,16 +260,21 @@ is CLI only, because it executes a command.
|
|
|
202
260
|
- It does not run your app, your judges or your eval. The agent (or a script you name
|
|
203
261
|
with `judge-run`) calls the judge model; the exported files run the eval in promptfoo,
|
|
204
262
|
DeepEval or Inspect AI.
|
|
205
|
-
-
|
|
206
|
-
pointwise
|
|
207
|
-
|
|
208
|
-
|
|
263
|
+
- A checked judge is wired in as the grader in promptfoo and DeepEval, and only a
|
|
264
|
+
pointwise one. promptfoo needs its prompt to answer in JSON (`{"pass": ...,
|
|
265
|
+
"reason": ...}`), because llm-rubric cannot parse a bare "pass". The DeepEval test
|
|
266
|
+
calls Ollama judges itself; for any other provider you fill in `call_judge`. The
|
|
267
|
+
Inspect AI export still uses Inspect's default `model_graded_qa` grader, not your
|
|
268
|
+
checked judge.
|
|
209
269
|
|
|
210
270
|
## Privacy and safety
|
|
211
271
|
|
|
212
272
|
Everything runs locally. eval-builder makes no network calls and sends nothing
|
|
213
273
|
anywhere. The only process it starts is the judge command you name with
|
|
214
|
-
`judge-run --enable-judge-plugin`.
|
|
274
|
+
`judge-run --enable-judge-plugin`. The labeling sheet is one HTML file with a content
|
|
275
|
+
security policy that blocks every network request; it keeps unexported labels in the
|
|
276
|
+
browser's local storage on that machine. Exported test files can call a model when
|
|
277
|
+
you run them (the DeepEval test calls the judge it was given). Redaction is on unless you pass `--no-redact`, and
|
|
215
278
|
the report lists what was replaced (counts and kinds, never the values).
|
|
216
279
|
|
|
217
280
|
## License
|
|
@@ -10,18 +10,26 @@ uvx eval-builder ingest sample_logs.jsonl # or your own logs: OpenAI, Anthro
|
|
|
10
10
|
uvx eval-builder select -n 10 --stratify category && uvx eval-builder draft
|
|
11
11
|
```
|
|
12
12
|
|
|
13
|
-
|
|
13
|
+

|
|
14
|
+
|
|
15
|
+
What you get (real output of the commands above, eval-builder 0.1.1 from PyPI; the
|
|
16
|
+
curl and the three uvx calls took 35 s in total here; the very first uvx run also
|
|
17
|
+
downloads about 38 MiB, mostly numpy, scipy and scikit-learn, which took 6 s here):
|
|
14
18
|
|
|
15
19
|
```
|
|
16
20
|
ingested 48 traces into evalset/traces.jsonl
|
|
17
21
|
sample_logs.jsonl: format=openai records=48 traces=48 skipped=0 sha256=2d5dd2295784988b
|
|
18
22
|
redactions: 0 {}
|
|
23
|
+
next: eval-builder select -n 30 (add --stratify <metadata keys> to cover them)
|
|
19
24
|
48 traces -> 16 unique (32 exact dupes, 0 near dupes) -> selected 10 (8 failures) across 4 clusters
|
|
20
25
|
q121-llama-13b: failure (negative user feedback); 69% of unique traces are failures and at least 30% of picks are reserved for them
|
|
26
|
+
q81-alpaca-13b: failure (negative user feedback); 69% of unique traces are failures and at least 30% of picks are reserved for them
|
|
21
27
|
q102-alpaca-13b: covers category=reasoning (2 unique traces, 12%)
|
|
22
28
|
q111-alpaca-13b: adds variety within cluster 1 (8 traces, 50%; triangle, response, person); least similar to cases already picked there
|
|
23
|
-
...
|
|
29
|
+
... (6 more picks)
|
|
30
|
+
next: run draft to turn the selection into cases.yaml
|
|
24
31
|
10 case(s) added, 10 total in evalset/cases.yaml
|
|
32
|
+
next: read the cases (list_cases, or cases.yaml), define criteria and judges (set_rubric, or rubric.yaml), fill expected_behavior and criteria per case and set status: ready (update_case), then run validate
|
|
25
33
|
```
|
|
26
34
|
|
|
27
35
|
`evalset/cases.yaml` now holds 10 real conversations with a TODO where the expected
|
|
@@ -71,7 +79,10 @@ export and report in 32 turns and about 10 minutes. From its final answer:
|
|
|
71
79
|
> Its verdicts will be unverified.
|
|
72
80
|
|
|
73
81
|
It wrote every expected behavior itself because nobody was there to confirm them, and
|
|
74
|
-
said so on each case. It did not invent human labels; it asked for them.
|
|
82
|
+
said so on each case. It did not invent human labels; it asked for them. That is where
|
|
83
|
+
this session and the earlier one both stopped, so 0.1.2 adds the missing step: `label`
|
|
84
|
+
writes a sheet the person labels in their browser, and `label import` brings the
|
|
85
|
+
labels back for judge-check ([Human labels, end to end](#human-labels-end-to-end)).
|
|
75
86
|
|
|
76
87
|
An earlier session against 0.1.0 is where the promptfoo context bug fixed in 0.1.1 came
|
|
77
88
|
from. The agent read the export and told the user: "Right now it only sends the final
|
|
@@ -131,6 +142,42 @@ The same check on the suite's own pass/fail judge (qwen2.5 7B, 24 cases, no huma
|
|
|
131
142
|
labels) gives `unstable`: its verdict changed across 5 identical calls on 7 of 24
|
|
132
143
|
cases (29%, interval 15% to 49%).
|
|
133
144
|
|
|
145
|
+
### Human labels, end to end
|
|
146
|
+
|
|
147
|
+
```sh
|
|
148
|
+
uvx eval-builder label # picks 24 cases, writes label_sheet.html
|
|
149
|
+
uvx eval-builder label import ~/Downloads/labels.jsonl # what the sheet's Export button saved
|
|
150
|
+
uvx eval-builder judge-check
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+

|
|
154
|
+
|
|
155
|
+
Run on the 48 sample conversations with every model's answer kept as a case (47
|
|
156
|
+
cases), three local judges (1,128 calls, 0 errors), and 24 labels entered through the
|
|
157
|
+
sheet in headless Chrome. The labels were made by the developer (Claude Code reading
|
|
158
|
+
each case for him), not by an independent annotator, so read this as a demonstration
|
|
159
|
+
of the flow. Details and every file: [`examples/sample-labeling/`](examples/sample-labeling/).
|
|
160
|
+
|
|
161
|
+
- `label` split the 24 picks 12/12 between cases the judges called pass and fail; 15
|
|
162
|
+
are cases where the judges disagree, flip or move under padding.
|
|
163
|
+
- The sheet made no network requests, survived a reload mid-way, and its download was
|
|
164
|
+
byte-identical to the copy box. `label import` took 24 rows, rejected 0.
|
|
165
|
+
- `judge-check` against those labels (fail 18, pass 6):
|
|
166
|
+
|
|
167
|
+
| judge | verdict | flip rate | accuracy vs labels | kappa |
|
|
168
|
+
|---|---|---|---|---|
|
|
169
|
+
| qwen2.5:7b-instruct, temp 0 | trustworthy | 0% [0%, 8%] | 75% [55%, 88%] | 0.50 [0.15, 0.85] |
|
|
170
|
+
| qwen2.5:7b-instruct, temp 0.8 | trustworthy | 9% [3%, 20%] | 71% [51%, 85%] | 0.44 [0.09, 0.79] |
|
|
171
|
+
| llama3.2:3b | unstable | 47% [33%, 61%] | 54% [35%, 72%] | 0.12 [-0.26, 0.50] |
|
|
172
|
+
|
|
173
|
+
The qwen judges pass the default thresholds, but judge-check also warns that their
|
|
174
|
+
accuracy is no better than always answering "fail" (75% of the labels), and every
|
|
175
|
+
miss went the same way: both passed a reply that said "Here is an allegorical poem"
|
|
176
|
+
and then wrote no poem. With 24 labels the kappa interval runs from about 0.1 to 0.8.
|
|
177
|
+
The DeepEval export wired the 0.8-temperature qwen judge in with its exact prompt;
|
|
178
|
+
DeepEval 4.2.8 ran three logged cases through it and it made the same call on the
|
|
179
|
+
missing poem.
|
|
180
|
+
|
|
134
181
|
## How it works
|
|
135
182
|
|
|
136
183
|
| step | what it does | what it uses |
|
|
@@ -140,14 +187,17 @@ cases (29%, interval 15% to 49%).
|
|
|
140
187
|
| `draft`, `validate` | Writes `cases.yaml` and `rubric.yaml` with TODO markers. The agent fills in expected behavior and criteria with you. Validation refuses ready cases that still contain TODO or reference unknown criteria. | PyYAML |
|
|
141
188
|
| `judge-plan` | Lists every judge call to make: each case N times, plus probes that swap the answer order (pairwise judges) and pad an answer with an irrelevant paragraph. | |
|
|
142
189
|
| `judge-run` | Optional and off by default. Sends each request as a JSON line to a command you name (your script, your provider, your keys) and records the verdicts. eval-builder ships no API keys and no provider code. | your command |
|
|
190
|
+
| `label`, `label import` | Picks the ready cases a person should label (default 24): the budget is split across outcomes (the judges' consensus, or the logged failure flag before judges ran) and up to half of each share goes to cases where judges disagree, flip across repeats or move under padding. Writes `label_sheet.html`, one offline file (one case per screen, pass/fail or A/B buttons, a note, keyboard shortcuts, judge verdicts hidden, progress kept in the browser) whose Export button downloads `labels.jsonl` in the format judge-check reads, and `label_sheet.csv` for spreadsheet users. `label import` checks the file against `cases.yaml` and merges it into the workspace. | Python standard library; the sheet is plain HTML and JavaScript |
|
|
143
191
|
| `judge-check` | Per judge: flip rate across repeated calls (with a Wilson interval), self-agreement, majority-of-3 vote stability, accuracy and Cohen's kappa against your human labels (with intervals), position consistency and first-shown preference, and how often padding moved the verdict toward the padded answer. Verdict: `trustworthy`, `unstable`, `biased`, `misaligned`, or `not_enough_data`, with the numbers behind it. | |
|
|
144
|
-
| `export` | promptfoo `promptfooconfig.yaml` (the full conversation as chat messages, llm-rubric asserts, and a judge that passed judge-check wired in as the grader when its rubric entry names a promptfoo `provider`), DeepEval dataset plus a `deepeval test run` file, Inspect AI dataset plus `task.py
|
|
192
|
+
| `export` | promptfoo `promptfooconfig.yaml` (the full conversation as chat messages, llm-rubric asserts, and a judge that passed judge-check wired in as the grader when its rubric entry names a promptfoo `provider`), DeepEval dataset plus a `deepeval test run` file (the same checked judge grades every case with its exact prompt; without one, GEval), Inspect AI dataset plus `task.py` (Inspect's default `model_graded_qa` grader), plain JSONL. The manifest lists file hashes and which judges passed. | |
|
|
145
193
|
| `report` | `report.md` and `report.json`: sources with sha256, counts, redactions, selection reasons, the judge table, and the limits. | |
|
|
146
194
|
|
|
147
195
|
The verdict thresholds are explicit flags with defaults: flip rate at most 20% of cases,
|
|
148
196
|
position consistency at least 80%, padding helps at most 10% of cases, kappa at least
|
|
149
197
|
0.4 on at least 20 human-labeled cases. A judge without human labels is never called
|
|
150
|
-
trustworthy.
|
|
198
|
+
trustworthy. When at least 80% of the selected cases or of the human labels share one
|
|
199
|
+
outcome, `select` and `judge-check` say so with the real proportions, because a judge
|
|
200
|
+
that always gives that answer would look accurate on them.
|
|
151
201
|
|
|
152
202
|
## Setup for agents
|
|
153
203
|
|
|
@@ -164,14 +214,22 @@ file it edits and does nothing on a second run. The workflow the agent follows i
|
|
|
164
214
|
[`skills/eval-builder/SKILL.md`](skills/eval-builder/SKILL.md).
|
|
165
215
|
|
|
166
216
|
MCP tools: `ingest`, `select`, `draft`, `list_cases`, `update_case`, `set_rubric`,
|
|
167
|
-
`validate`, `judge_plan`, `judge_check`, `export`, `report`,
|
|
217
|
+
`validate`, `judge_plan`, `label`, `label_import`, `judge_check`, `export`, `report`,
|
|
218
|
+
`status`. The judge runner
|
|
168
219
|
is CLI only, because it executes a command.
|
|
169
220
|
|
|
170
221
|
## What it can't do
|
|
171
222
|
|
|
172
223
|
- It does not write expected behavior or human labels. The agent drafts expected
|
|
173
224
|
behavior with you; labels must come from people. Without labels, judge-check can
|
|
174
|
-
tell you a judge is unstable or biased, but not that it is right.
|
|
225
|
+
tell you a judge is unstable or biased, but not that it is right. `label` makes the
|
|
226
|
+
labeling quick, but someone still has to read each case.
|
|
227
|
+
- Picking labels where judges disagree makes each label more informative, but those
|
|
228
|
+
cases are harder than average, so accuracy measured on them leans pessimistic.
|
|
229
|
+
`--uncertain-share 0` picks by outcome and topic only.
|
|
230
|
+
- The labeling sheet is a static page, so it cannot save files: the person has to
|
|
231
|
+
click Export (or copy the text) and import it. Unexported progress lives only in
|
|
232
|
+
that browser's local storage.
|
|
175
233
|
- Selection is lexical. Two requests that mean the same thing in different words can
|
|
176
234
|
land in different clusters, and near-duplicate detection only catches close textual
|
|
177
235
|
matches.
|
|
@@ -183,16 +241,21 @@ is CLI only, because it executes a command.
|
|
|
183
241
|
- It does not run your app, your judges or your eval. The agent (or a script you name
|
|
184
242
|
with `judge-run`) calls the judge model; the exported files run the eval in promptfoo,
|
|
185
243
|
DeepEval or Inspect AI.
|
|
186
|
-
-
|
|
187
|
-
pointwise
|
|
188
|
-
|
|
189
|
-
|
|
244
|
+
- A checked judge is wired in as the grader in promptfoo and DeepEval, and only a
|
|
245
|
+
pointwise one. promptfoo needs its prompt to answer in JSON (`{"pass": ...,
|
|
246
|
+
"reason": ...}`), because llm-rubric cannot parse a bare "pass". The DeepEval test
|
|
247
|
+
calls Ollama judges itself; for any other provider you fill in `call_judge`. The
|
|
248
|
+
Inspect AI export still uses Inspect's default `model_graded_qa` grader, not your
|
|
249
|
+
checked judge.
|
|
190
250
|
|
|
191
251
|
## Privacy and safety
|
|
192
252
|
|
|
193
253
|
Everything runs locally. eval-builder makes no network calls and sends nothing
|
|
194
254
|
anywhere. The only process it starts is the judge command you name with
|
|
195
|
-
`judge-run --enable-judge-plugin`.
|
|
255
|
+
`judge-run --enable-judge-plugin`. The labeling sheet is one HTML file with a content
|
|
256
|
+
security policy that blocks every network request; it keeps unexported labels in the
|
|
257
|
+
browser's local storage on that machine. Exported test files can call a model when
|
|
258
|
+
you run them (the DeepEval test calls the judge it was given). Redaction is on unless you pass `--no-redact`, and
|
|
196
259
|
the report lists what was replaced (counts and kinds, never the values).
|
|
197
260
|
|
|
198
261
|
## License
|
|
Binary file
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
Output docs/demo.gif
|
|
2
|
+
Set FontSize 14
|
|
3
|
+
Set Width 1600
|
|
4
|
+
Set Height 900
|
|
5
|
+
Set Theme "Builtin Dark"
|
|
6
|
+
Set TypingSpeed 30ms
|
|
7
|
+
Hide
|
|
8
|
+
Type "cd \"$(mktemp -d)\" && clear"
|
|
9
|
+
Enter
|
|
10
|
+
Show
|
|
11
|
+
Type "curl -sLO https://raw.githubusercontent.com/Abelo9996/eval-builder/main/examples/sample_logs.jsonl"
|
|
12
|
+
Enter
|
|
13
|
+
Wait
|
|
14
|
+
Type "uvx eval-builder ingest sample_logs.jsonl"
|
|
15
|
+
Enter
|
|
16
|
+
Wait
|
|
17
|
+
Sleep 1.5s
|
|
18
|
+
Type "uvx eval-builder select -n 10 --stratify category"
|
|
19
|
+
Enter
|
|
20
|
+
Wait+Screen@20s /next: run draft/
|
|
21
|
+
Sleep 2s
|
|
22
|
+
Type "uvx eval-builder draft && grep -m1 -A10 'id: case-002' evalset/cases.yaml | cut -c1-120"
|
|
23
|
+
Enter
|
|
24
|
+
Wait+Screen@20s /role: user/
|
|
25
|
+
Sleep 5s
|
|
Binary file
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# Labeling example
|
|
2
|
+
|
|
3
|
+
The human-labeling step, end to end, on the 48 conversations in
|
|
4
|
+
[`../sample_logs.jsonl`](../sample_logs.jsonl) (16 MT-Bench questions, each answered
|
|
5
|
+
by gpt-4, alpaca-13b and llama-13b; CC BY 4.0). Everything here was produced by
|
|
6
|
+
[`run.sh`](run.sh) on an Apple M4 MacBook (16 GB) on October 8 and 9, 2026, with
|
|
7
|
+
eval-builder 0.1.2 from this repository.
|
|
8
|
+
|
|
9
|
+
**Who labeled.** The labels in [`evalset/labels.jsonl`](evalset/labels.jsonl) were made
|
|
10
|
+
by the developer, not by an independent annotator: Claude Code, working for the
|
|
11
|
+
developer, read each picked case in full (earlier turns, reply, expected behavior) and
|
|
12
|
+
recorded a decision with a note for the close calls ([`decisions.json`](decisions.json)).
|
|
13
|
+
[`drive_sheet.py`](drive_sheet.py) then entered those decisions through the real
|
|
14
|
+
`label_sheet.html` in headless Chrome, using the keyboard shortcuts, and clicked Export.
|
|
15
|
+
The expected behavior in [`fill.py`](fill.py) was written the same way. Treat the
|
|
16
|
+
numbers below as a demonstration of the flow, not as ground truth about these judges.
|
|
17
|
+
|
|
18
|
+
## Steps and what they printed
|
|
19
|
+
|
|
20
|
+
1. `select --dedupe-on input+output --near-dup 0.99` kept each model's answer: 48
|
|
21
|
+
traces, 47 unique (two answers to one question were near-identical). All 47 became
|
|
22
|
+
ready cases with one expected behavior per question.
|
|
23
|
+
2. Three local judges through Ollama (`examples/judges/ollama_judge.py`), one pass/fail
|
|
24
|
+
prompt that answers in JSON, 5 trials per case plus 3 with an irrelevant paragraph
|
|
25
|
+
appended to the reply: 1,128 calls, 0 errors (qwen2.5:7b-instruct at temperature
|
|
26
|
+
0.8: 1,114 s; the same model at temperature 0: 1,130 s; llama3.2:3b: 589 s; the
|
|
27
|
+
machine was shared with other jobs).
|
|
28
|
+
3. `judge-check` without labels: llama3.2-3b `unstable` (verdict changed across repeats
|
|
29
|
+
on 47% of cases), both qwen judges `not_enough_data` ("stable, but only 0 case(s)
|
|
30
|
+
have human labels").
|
|
31
|
+
4. `label` picked 24 of the 47 cases: 12 where the judges' consensus was fail and 12
|
|
32
|
+
where it was pass; 15 of the 24 are cases where judges disagree, flip or move under
|
|
33
|
+
padding ([`evalset/label_plan.json`](evalset/label_plan.json) has every reason).
|
|
34
|
+
5. The sheet in headless Chrome ([`sheet-run/drive_log.txt`](sheet-run/drive_log.txt)):
|
|
35
|
+
24 cases, a reload after 13 labels kept "13 of 24 labeled", the download
|
|
36
|
+
(`labels.jsonl`, 4,292 bytes) was identical to the copy box, 2 requests in total,
|
|
37
|
+
none of them to anything but `file:`, `blob:` or `data:` URLs, no console errors. Screenshots:
|
|
38
|
+
[first case](sheet-run/sheet-first-case.png), [phone width](sheet-run/sheet-phone.png),
|
|
39
|
+
[export](sheet-run/sheet-export.png).
|
|
40
|
+
6. `label import`: 24 imported, 0 rejected; labels fail 18, pass 6.
|
|
41
|
+
7. `judge-check` with the labels ([`evalset/report.md`](evalset/report.md)):
|
|
42
|
+
|
|
43
|
+
| judge | verdict | flip rate | accuracy vs labels | kappa | padding helped |
|
|
44
|
+
|---|---|---|---|---|---|
|
|
45
|
+
| qwen2.5:7b-instruct, temp 0 | trustworthy | 0% [0%, 8%] | 75% [55%, 88%] | 0.50 [0.15, 0.85] | 0% [0%, 8%] |
|
|
46
|
+
| qwen2.5:7b-instruct, temp 0.8 | trustworthy | 9% [3%, 20%] | 71% [51%, 85%] | 0.44 [0.09, 0.79] | 0% [0%, 8%] |
|
|
47
|
+
| llama3.2:3b | unstable | 47% [33%, 61%] | 54% [35%, 72%] | 0.12 [-0.26, 0.50] | 4% [1%, 14%] |
|
|
48
|
+
|
|
49
|
+
n = 47 cases for flip rate and padding, 24 labeled cases for accuracy and kappa;
|
|
50
|
+
brackets are 95% intervals.
|
|
51
|
+
|
|
52
|
+
## What it shows, read plainly
|
|
53
|
+
|
|
54
|
+
- Both qwen judges clear the default thresholds (kappa at least 0.4 on at least 20
|
|
55
|
+
labels), so the export wires `qwen2.5-7b` (the first that passed) into promptfoo and
|
|
56
|
+
DeepEval. The kappa intervals are wide, from about 0.1 to 0.8: 24 labels cannot pin
|
|
57
|
+
the agreement down.
|
|
58
|
+
- Their accuracy (71% and 75%) is no better than always answering "fail" (75% of the
|
|
59
|
+
labels are fail), and judge-check now says so in a warning. Every miss went the same
|
|
60
|
+
way: the judge passed a reply the labels failed. Both qwen judges passed case-018
|
|
61
|
+
(llama-13b answered "Here is an allegorical poem that illustrates the above:" and
|
|
62
|
+
no poem) and case-019 (alpaca-13b's rewrite in which most sentences do not start with
|
|
63
|
+
"A").
|
|
64
|
+
- The labels came out 18 fail to 6 pass even though `label` split the picks 12/12 by
|
|
65
|
+
the judges' consensus: the judges said pass far more often than the labels did.
|
|
66
|
+
- llama3.2:3b changes its verdict on almost half the cases between identical calls.
|
|
67
|
+
|
|
68
|
+
## DeepEval with the checked judge
|
|
69
|
+
|
|
70
|
+
`export` wrote `evalset/exports/deepeval/judge.json` (judge `qwen2.5-7b`, its exact
|
|
71
|
+
prompt, provider `ollama:chat:qwen2.5:7b-instruct` at temperature 0.8). DeepEval 4.2.8
|
|
72
|
+
ran three of the cases on the logged outputs
|
|
73
|
+
(`EVAL_BUILDER_USE_OBSERVED=1 pytest test_eval_builder.py -k "case-018 or case-020 or case-040"`):
|
|
74
|
+
case-020 passed, case-040 failed with the judge's reason
|
|
75
|
+
`{"pass": false, "reason": "Incorrect circumradius calculation."}`, and case-018 (the
|
|
76
|
+
missing poem) passed, the same mistake judge-check found.
|
|
77
|
+
|
|
78
|
+
`evalset/judge_requests.jsonl` (4.7 MB of rendered prompts) is not committed; `judge-plan`
|
|
79
|
+
rebuilds it from `cases.yaml` and `rubric.yaml`.
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
{
|
|
2
|
+
"case-006": {
|
|
3
|
+
"label": "fail",
|
|
4
|
+
"note": ""
|
|
5
|
+
},
|
|
6
|
+
"case-007": {
|
|
7
|
+
"label": "fail",
|
|
8
|
+
"note": "same textbook definitions, not for a five-year-old"
|
|
9
|
+
},
|
|
10
|
+
"case-008": {
|
|
11
|
+
"label": "pass",
|
|
12
|
+
"note": ""
|
|
13
|
+
},
|
|
14
|
+
"case-009": {
|
|
15
|
+
"label": "fail",
|
|
16
|
+
"note": "repeats one definition, not simplified"
|
|
17
|
+
},
|
|
18
|
+
"case-010": {
|
|
19
|
+
"label": "fail",
|
|
20
|
+
"note": "one sentence, no program"
|
|
21
|
+
},
|
|
22
|
+
"case-012": {
|
|
23
|
+
"label": "fail",
|
|
24
|
+
"note": ""
|
|
25
|
+
},
|
|
26
|
+
"case-013": {
|
|
27
|
+
"label": "fail",
|
|
28
|
+
"note": "wrong start values and sums two terms, not three"
|
|
29
|
+
},
|
|
30
|
+
"case-015": {
|
|
31
|
+
"label": "fail",
|
|
32
|
+
"note": "Fibonacci-style, wrong recurrence"
|
|
33
|
+
},
|
|
34
|
+
"case-016": {
|
|
35
|
+
"label": "fail",
|
|
36
|
+
"note": "names the stages literally; not an allegory"
|
|
37
|
+
},
|
|
38
|
+
"case-018": {
|
|
39
|
+
"label": "fail",
|
|
40
|
+
"note": "no poem"
|
|
41
|
+
},
|
|
42
|
+
"case-019": {
|
|
43
|
+
"label": "fail",
|
|
44
|
+
"note": "most sentences do not start with A"
|
|
45
|
+
},
|
|
46
|
+
"case-023": {
|
|
47
|
+
"label": "pass",
|
|
48
|
+
"note": "names one improvement; misses that the email is not short"
|
|
49
|
+
},
|
|
50
|
+
"case-024": {
|
|
51
|
+
"label": "fail",
|
|
52
|
+
"note": "answers a different message"
|
|
53
|
+
},
|
|
54
|
+
"case-025": {
|
|
55
|
+
"label": "fail",
|
|
56
|
+
"note": "'US President' and 'Lewis' are not specific people"
|
|
57
|
+
},
|
|
58
|
+
"case-027": {
|
|
59
|
+
"label": "fail",
|
|
60
|
+
"note": "three lines, Washington instead of Roosevelt, da Vinci under science"
|
|
61
|
+
},
|
|
62
|
+
"case-029": {
|
|
63
|
+
"label": "pass",
|
|
64
|
+
"note": "explains why it is a contradiction"
|
|
65
|
+
},
|
|
66
|
+
"case-031": {
|
|
67
|
+
"label": "pass",
|
|
68
|
+
"note": "says no, without explaining why"
|
|
69
|
+
},
|
|
70
|
+
"case-032": {
|
|
71
|
+
"label": "pass",
|
|
72
|
+
"note": ""
|
|
73
|
+
},
|
|
74
|
+
"case-034": {
|
|
75
|
+
"label": "fail",
|
|
76
|
+
"note": "says $4000, should be $2000"
|
|
77
|
+
},
|
|
78
|
+
"case-036": {
|
|
79
|
+
"label": "fail",
|
|
80
|
+
"note": "ratings reordered and only one date"
|
|
81
|
+
},
|
|
82
|
+
"case-039": {
|
|
83
|
+
"label": "fail",
|
|
84
|
+
"note": "12 is wrong; 5*pi"
|
|
85
|
+
},
|
|
86
|
+
"case-042": {
|
|
87
|
+
"label": "fail",
|
|
88
|
+
"note": "restates definitions, no real assumptions examined"
|
|
89
|
+
},
|
|
90
|
+
"case-043": {
|
|
91
|
+
"label": "pass",
|
|
92
|
+
"note": ""
|
|
93
|
+
},
|
|
94
|
+
"case-047": {
|
|
95
|
+
"label": "fail",
|
|
96
|
+
"note": "repeats a wrong claim"
|
|
97
|
+
}
|
|
98
|
+
}
|