eval-builder 0.1.1__tar.gz → 0.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (121) hide show
  1. {eval_builder-0.1.1 → eval_builder-0.1.2}/.gitignore +2 -0
  2. {eval_builder-0.1.1 → eval_builder-0.1.2}/AGENTS.md +5 -1
  3. {eval_builder-0.1.1 → eval_builder-0.1.2}/CHANGELOG.md +40 -0
  4. {eval_builder-0.1.1 → eval_builder-0.1.2}/PKG-INFO +76 -13
  5. {eval_builder-0.1.1 → eval_builder-0.1.2}/README.md +75 -12
  6. eval_builder-0.1.2/docs/demo.gif +0 -0
  7. eval_builder-0.1.2/docs/demo.tape +25 -0
  8. eval_builder-0.1.2/docs/label-sheet.png +0 -0
  9. eval_builder-0.1.2/examples/sample-labeling/README.md +79 -0
  10. eval_builder-0.1.2/examples/sample-labeling/decisions.json +98 -0
  11. eval_builder-0.1.2/examples/sample-labeling/drive_sheet.py +90 -0
  12. eval_builder-0.1.2/examples/sample-labeling/evalset/cases.yaml +2042 -0
  13. eval_builder-0.1.2/examples/sample-labeling/evalset/exports/deepeval/dataset.json +1145 -0
  14. eval_builder-0.1.2/examples/sample-labeling/evalset/exports/deepeval/judge.json +10 -0
  15. eval_builder-0.1.2/examples/sample-labeling/evalset/exports/deepeval/test_eval_builder.py +165 -0
  16. eval_builder-0.1.2/examples/sample-labeling/evalset/exports/inspect/dataset.jsonl +47 -0
  17. eval_builder-0.1.2/examples/sample-labeling/evalset/exports/inspect/task.py +23 -0
  18. eval_builder-0.1.2/examples/sample-labeling/evalset/exports/jsonl/cases.jsonl +47 -0
  19. eval_builder-0.1.2/examples/sample-labeling/evalset/exports/manifest.json +90 -0
  20. eval_builder-0.1.2/examples/sample-labeling/evalset/exports/promptfoo/promptfooconfig.yaml +2314 -0
  21. eval_builder-0.1.2/examples/sample-labeling/evalset/ingest.json +33 -0
  22. eval_builder-0.1.2/examples/sample-labeling/evalset/judge_check.json +2188 -0
  23. eval_builder-0.1.2/examples/sample-labeling/evalset/judge_run_log.json +35 -0
  24. eval_builder-0.1.2/examples/sample-labeling/evalset/judgments.jsonl +1128 -0
  25. eval_builder-0.1.2/examples/sample-labeling/evalset/label_plan.json +673 -0
  26. eval_builder-0.1.2/examples/sample-labeling/evalset/label_sheet.csv +313 -0
  27. eval_builder-0.1.2/examples/sample-labeling/evalset/label_sheet.html +445 -0
  28. eval_builder-0.1.2/examples/sample-labeling/evalset/labels.jsonl +24 -0
  29. eval_builder-0.1.2/examples/sample-labeling/evalset/report.json +3593 -0
  30. eval_builder-0.1.2/examples/sample-labeling/evalset/report.md +148 -0
  31. eval_builder-0.1.2/examples/sample-labeling/evalset/rubric.yaml +100 -0
  32. eval_builder-0.1.2/examples/sample-labeling/evalset/selection.json +1227 -0
  33. eval_builder-0.1.2/examples/sample-labeling/evalset/traces.jsonl +48 -0
  34. eval_builder-0.1.2/examples/sample-labeling/fill.py +127 -0
  35. eval_builder-0.1.2/examples/sample-labeling/run.sh +35 -0
  36. eval_builder-0.1.2/examples/sample-labeling/sheet-run/drive_log.txt +12 -0
  37. eval_builder-0.1.2/examples/sample-labeling/sheet-run/labels.jsonl +24 -0
  38. eval_builder-0.1.2/examples/sample-labeling/sheet-run/sheet-export.png +0 -0
  39. eval_builder-0.1.2/examples/sample-labeling/sheet-run/sheet-first-case.png +0 -0
  40. eval_builder-0.1.2/examples/sample-labeling/sheet-run/sheet-phone.png +0 -0
  41. {eval_builder-0.1.1 → eval_builder-0.1.2}/pyproject.toml +5 -1
  42. {eval_builder-0.1.1 → eval_builder-0.1.2}/skills/eval-builder/SKILL.md +28 -13
  43. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/__init__.py +1 -1
  44. eval_builder-0.1.2/src/eval_builder/balance.py +55 -0
  45. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/cli.py +88 -2
  46. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/export.py +199 -22
  47. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/judge/check.py +69 -10
  48. eval_builder-0.1.2/src/eval_builder/label.py +549 -0
  49. eval_builder-0.1.2/src/eval_builder/label_sheet.py +481 -0
  50. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/mcp_server.py +43 -9
  51. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/report.py +14 -0
  52. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/select.py +24 -0
  53. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/status.py +9 -0
  54. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/workspace.py +12 -0
  55. eval_builder-0.1.2/tests/test_label.py +349 -0
  56. {eval_builder-0.1.1 → eval_builder-0.1.2}/uv.lock +1 -1
  57. {eval_builder-0.1.1 → eval_builder-0.1.2}/.github/workflows/ci.yml +0 -0
  58. {eval_builder-0.1.1 → eval_builder-0.1.2}/.github/workflows/release.yml +0 -0
  59. {eval_builder-0.1.1 → eval_builder-0.1.2}/CONTRIBUTING.md +0 -0
  60. {eval_builder-0.1.1 → eval_builder-0.1.2}/LICENSE +0 -0
  61. {eval_builder-0.1.1 → eval_builder-0.1.2}/SECURITY.md +0 -0
  62. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/judges/control_judges.py +0 -0
  63. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/judges/ollama_judge.py +0 -0
  64. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/README.md +0 -0
  65. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/expected_behaviors.yaml +0 -0
  66. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/fill_suite.py +0 -0
  67. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/judges/cases.yaml +0 -0
  68. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/judges/judge_check.json +0 -0
  69. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/judges/judge_run_log.json +0 -0
  70. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/judges/judgments.jsonl +0 -0
  71. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/judges/labels.jsonl +0 -0
  72. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/judges/report.json +0 -0
  73. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/judges/report.md +0 -0
  74. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/judges/rubric.yaml +0 -0
  75. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/prepare.py +0 -0
  76. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/run.sh +0 -0
  77. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/cases.yaml +0 -0
  78. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/exports/deepeval/dataset.json +0 -0
  79. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/exports/deepeval/test_eval_builder.py +0 -0
  80. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/exports/inspect/dataset.jsonl +0 -0
  81. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/exports/inspect/task.py +0 -0
  82. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/exports/jsonl/cases.jsonl +0 -0
  83. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/exports/manifest.json +0 -0
  84. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/exports/promptfoo/promptfooconfig.yaml +0 -0
  85. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/ingest.json +0 -0
  86. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/judge_check.json +0 -0
  87. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/judge_run_log.json +0 -0
  88. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/judgments.jsonl +0 -0
  89. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/report.json +0 -0
  90. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/report.md +0 -0
  91. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/rubric.yaml +0 -0
  92. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/suite/selection.json +0 -0
  93. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/mt-bench/verify_exports.sh +0 -0
  94. {eval_builder-0.1.1 → eval_builder-0.1.2}/examples/sample_logs.jsonl +0 -0
  95. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/draft.py +0 -0
  96. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/ingest/__init__.py +0 -0
  97. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/ingest/formats.py +0 -0
  98. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/io.py +0 -0
  99. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/judge/__init__.py +0 -0
  100. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/judge/plan.py +0 -0
  101. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/judge/run.py +0 -0
  102. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/judge/stats.py +0 -0
  103. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/redact.py +0 -0
  104. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/schema.py +0 -0
  105. {eval_builder-0.1.1 → eval_builder-0.1.2}/src/eval_builder/setup_agents.py +0 -0
  106. {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/conftest.py +0 -0
  107. {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/fixtures/anthropic_messages.json +0 -0
  108. {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/fixtures/generic.jsonl +0 -0
  109. {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/fixtures/langfuse_export.json +0 -0
  110. {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/fixtures/openai_chat.jsonl +0 -0
  111. {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/fixtures/otel_genai.json +0 -0
  112. {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_cli_mcp.py +0 -0
  113. {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_draft.py +0 -0
  114. {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_export.py +0 -0
  115. {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_ingest.py +0 -0
  116. {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_judge_check.py +0 -0
  117. {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_judge_run.py +0 -0
  118. {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_judge_stats.py +0 -0
  119. {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_report.py +0 -0
  120. {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_select.py +0 -0
  121. {eval_builder-0.1.1 → eval_builder-0.1.2}/tests/test_setup.py +0 -0
@@ -9,9 +9,11 @@ build/
9
9
  *.egg-info/
10
10
  .DS_Store
11
11
  evalset/
12
+ !examples/sample-labeling/evalset/
12
13
  .deepeval/
13
14
  # Example inputs that prepare.py downloads or derives (reproducible from the pinned dataset)
14
15
  examples/mt-bench/data/
15
16
  # Large or derived files inside example workspaces
16
17
  examples/mt-bench/*/judge_requests.jsonl
17
18
  examples/mt-bench/*/traces.jsonl
19
+ examples/sample-labeling/evalset/judge_requests.jsonl
@@ -16,6 +16,9 @@ src/eval_builder/
16
16
  judge/run.py opt-in judge plugin runner (off by default, runs a user command)
17
17
  judge/check.py flip rate, kappa, accuracy, probes, verdicts
18
18
  judge/stats.py Wilson interval, Cohen's kappa, majority vote
19
+ label.py which cases a person should label, the CSV sheet, `label import`
20
+ label_sheet.py the offline HTML labeling sheet (one file, inline JS, no network)
21
+ balance.py the warning when cases or labels are mostly one outcome
19
22
  export.py promptfoo, DeepEval, Inspect AI, JSONL
20
23
  report.py report.md and report.json
21
24
  setup_agents.py `eval-builder setup` for Claude Code, Codex, Cursor
@@ -25,7 +28,8 @@ src/eval_builder/
25
28
 
26
29
  ## Rules
27
30
 
28
- - No model calls and no network access in the package. The judge runner only starts a
31
+ - No model calls and no network access in the package. (Exported files may call a
32
+ model when the user runs them, for example the DeepEval test calling a wired judge.) The judge runner only starts a
29
33
  command the user names, and only with an explicit flag.
30
34
  - Deterministic: same inputs and seed give the same selection and the same files.
31
35
  - Every number the tool reports must be traceable to a file in the workspace.
@@ -1,5 +1,45 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.1.2 (2026-10-08)
4
+
5
+ Human labels. No judge can be called trustworthy without them, and both real agent
6
+ sessions against 0.1.0 and 0.1.1 stopped at that step: they asked for labels and had
7
+ no way to collect them.
8
+
9
+ - `label` (CLI, and MCP `label`): picks the ready cases a person should label, 24 by
10
+ default (judge-check needs 20). The budget is split evenly across outcomes (the
11
+ judges' consensus verdict, or the logged failure flag before any judge has run), and
12
+ up to half of each share goes to cases where the judges disagree, flip across
13
+ repeats or move under padding. Every pick and its reason is in `label_plan.json`.
14
+ - `label` writes `label_sheet.html`: one self-contained file that works offline and
15
+ makes no network requests (its content security policy blocks them). One case per
16
+ screen with the earlier turns, the reply, the expected behavior and criteria;
17
+ pass/fail or A/B buttons (labels come from rubric.yaml), an optional note, keyboard
18
+ shortcuts, progress, and the judges' verdicts hidden. Progress survives a reload
19
+ through the browser's local storage. Export downloads `labels.jsonl` in exactly the
20
+ format judge-check reads and shows the same text to copy.
21
+ - `label` also writes `label_sheet.csv` for spreadsheet users (fill the label column).
22
+ - `label import <file>` (CLI, and MCP `label_import`): reads the sheet's labels.jsonl,
23
+ the filled-in CSV, a JSON list, or stdin (`-`); checks every case id and label
24
+ against cases.yaml and rubric.yaml; merges into `labels.jsonl` (a new label replaces
25
+ the same labeler's earlier one, other labelers are kept); lists rejected rows with
26
+ the reason and exits 1 when there are any.
27
+ - `select` and `judge-check` warn, with the real proportions, when at least 80% of the
28
+ selected cases or of the human labels share one outcome. judge-check also warns when
29
+ a judge gave the same verdict on every labeled case or its accuracy is no better
30
+ than always giving the most common label, and reports `majority_baseline` (that
31
+ always-the-common-label accuracy).
32
+ - DeepEval export: a pointwise judge that passed judge-check (or one forced with
33
+ `--judge`) now grades every case with its exact rubric.yaml prompt, through a custom
34
+ metric in `test_eval_builder.py` and `judge.json`. Ollama providers are called on the
35
+ local server; for other providers you fill in `call_judge`. Without a checked judge
36
+ the file keeps GEval. The export notes and README say that Inspect AI still uses its
37
+ default `model_graded_qa` grader.
38
+ - judge-check parses JSON verdicts that a token limit cut off mid-reason, and verdicts
39
+ wrapped in a json code fence.
40
+ - `status` and judge-check's `next` walk through the labeling step.
41
+ - `examples/sample-labeling/`: the whole flow on the bundled 48-conversation sample.
42
+
3
43
  ## 0.1.1 (2026-10-08)
4
44
 
5
45
  Fixes from a fresh-install audit and a real Claude Code session driving the MCP server.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: eval-builder
3
- Version: 0.1.1
3
+ Version: 0.1.2
4
4
  Summary: Turn real LLM app logs into an eval suite and measure which LLM judges you can trust.
5
5
  Author: Abel Yagubyan
6
6
  License-Expression: MIT
@@ -29,18 +29,26 @@ uvx eval-builder ingest sample_logs.jsonl # or your own logs: OpenAI, Anthro
29
29
  uvx eval-builder select -n 10 --stratify category && uvx eval-builder draft
30
30
  ```
31
31
 
32
- What you get (real output, eval-builder 0.1.1 on the sample file):
32
+ ![The quickstart running against eval-builder 0.1.1 from PyPI](docs/demo.gif)
33
+
34
+ What you get (real output of the commands above, eval-builder 0.1.1 from PyPI; the
35
+ curl and the three uvx calls took 35 s in total here; the very first uvx run also
36
+ downloads about 38 MiB, mostly numpy, scipy and scikit-learn, which took 6 s here):
33
37
 
34
38
  ```
35
39
  ingested 48 traces into evalset/traces.jsonl
36
40
  sample_logs.jsonl: format=openai records=48 traces=48 skipped=0 sha256=2d5dd2295784988b
37
41
  redactions: 0 {}
42
+ next: eval-builder select -n 30 (add --stratify <metadata keys> to cover them)
38
43
  48 traces -> 16 unique (32 exact dupes, 0 near dupes) -> selected 10 (8 failures) across 4 clusters
39
44
  q121-llama-13b: failure (negative user feedback); 69% of unique traces are failures and at least 30% of picks are reserved for them
45
+ q81-alpaca-13b: failure (negative user feedback); 69% of unique traces are failures and at least 30% of picks are reserved for them
40
46
  q102-alpaca-13b: covers category=reasoning (2 unique traces, 12%)
41
47
  q111-alpaca-13b: adds variety within cluster 1 (8 traces, 50%; triangle, response, person); least similar to cases already picked there
42
- ...
48
+ ... (6 more picks)
49
+ next: run draft to turn the selection into cases.yaml
43
50
  10 case(s) added, 10 total in evalset/cases.yaml
51
+ next: read the cases (list_cases, or cases.yaml), define criteria and judges (set_rubric, or rubric.yaml), fill expected_behavior and criteria per case and set status: ready (update_case), then run validate
44
52
  ```
45
53
 
46
54
  `evalset/cases.yaml` now holds 10 real conversations with a TODO where the expected
@@ -90,7 +98,10 @@ export and report in 32 turns and about 10 minutes. From its final answer:
90
98
  > Its verdicts will be unverified.
91
99
 
92
100
  It wrote every expected behavior itself because nobody was there to confirm them, and
93
- said so on each case. It did not invent human labels; it asked for them.
101
+ said so on each case. It did not invent human labels; it asked for them. That is where
102
+ this session and the earlier one both stopped, so 0.1.2 adds the missing step: `label`
103
+ writes a sheet the person labels in their browser, and `label import` brings the
104
+ labels back for judge-check ([Human labels, end to end](#human-labels-end-to-end)).
94
105
 
95
106
  An earlier session against 0.1.0 is where the promptfoo context bug fixed in 0.1.1 came
96
107
  from. The agent read the export and told the user: "Right now it only sends the final
@@ -150,6 +161,42 @@ The same check on the suite's own pass/fail judge (qwen2.5 7B, 24 cases, no huma
150
161
  labels) gives `unstable`: its verdict changed across 5 identical calls on 7 of 24
151
162
  cases (29%, interval 15% to 49%).
152
163
 
164
+ ### Human labels, end to end
165
+
166
+ ```sh
167
+ uvx eval-builder label # picks 24 cases, writes label_sheet.html
168
+ uvx eval-builder label import ~/Downloads/labels.jsonl # what the sheet's Export button saved
169
+ uvx eval-builder judge-check
170
+ ```
171
+
172
+ ![The labeling sheet: one case per screen, pass and fail buttons with keyboard shortcuts](docs/label-sheet.png)
173
+
174
+ Run on the 48 sample conversations with every model's answer kept as a case (47
175
+ cases), three local judges (1,128 calls, 0 errors), and 24 labels entered through the
176
+ sheet in headless Chrome. The labels were made by the developer (Claude Code reading
177
+ each case for him), not by an independent annotator, so read this as a demonstration
178
+ of the flow. Details and every file: [`examples/sample-labeling/`](examples/sample-labeling/).
179
+
180
+ - `label` split the 24 picks 12/12 between cases the judges called pass and fail; 15
181
+ are cases where the judges disagree, flip or move under padding.
182
+ - The sheet made no network requests, survived a reload mid-way, and its download was
183
+ byte-identical to the copy box. `label import` took 24 rows, rejected 0.
184
+ - `judge-check` against those labels (fail 18, pass 6):
185
+
186
+ | judge | verdict | flip rate | accuracy vs labels | kappa |
187
+ |---|---|---|---|---|
188
+ | qwen2.5:7b-instruct, temp 0 | trustworthy | 0% [0%, 8%] | 75% [55%, 88%] | 0.50 [0.15, 0.85] |
189
+ | qwen2.5:7b-instruct, temp 0.8 | trustworthy | 9% [3%, 20%] | 71% [51%, 85%] | 0.44 [0.09, 0.79] |
190
+ | llama3.2:3b | unstable | 47% [33%, 61%] | 54% [35%, 72%] | 0.12 [-0.26, 0.50] |
191
+
192
+ The qwen judges pass the default thresholds, but judge-check also warns that their
193
+ accuracy is no better than always answering "fail" (75% of the labels), and every
194
+ miss went the same way: both passed a reply that said "Here is an allegorical poem"
195
+ and then wrote no poem. With 24 labels the kappa interval runs from about 0.1 to 0.8.
196
+ The DeepEval export wired the 0.8-temperature qwen judge in with its exact prompt;
197
+ DeepEval 4.2.8 ran three logged cases through it and it made the same call on the
198
+ missing poem.
199
+
153
200
  ## How it works
154
201
 
155
202
  | step | what it does | what it uses |
@@ -159,14 +206,17 @@ cases (29%, interval 15% to 49%).
159
206
  | `draft`, `validate` | Writes `cases.yaml` and `rubric.yaml` with TODO markers. The agent fills in expected behavior and criteria with you. Validation refuses ready cases that still contain TODO or reference unknown criteria. | PyYAML |
160
207
  | `judge-plan` | Lists every judge call to make: each case N times, plus probes that swap the answer order (pairwise judges) and pad an answer with an irrelevant paragraph. | |
161
208
  | `judge-run` | Optional and off by default. Sends each request as a JSON line to a command you name (your script, your provider, your keys) and records the verdicts. eval-builder ships no API keys and no provider code. | your command |
209
+ | `label`, `label import` | Picks the ready cases a person should label (default 24): the budget is split across outcomes (the judges' consensus, or the logged failure flag before judges ran) and up to half of each share goes to cases where judges disagree, flip across repeats or move under padding. Writes `label_sheet.html`, one offline file (one case per screen, pass/fail or A/B buttons, a note, keyboard shortcuts, judge verdicts hidden, progress kept in the browser) whose Export button downloads `labels.jsonl` in the format judge-check reads, and `label_sheet.csv` for spreadsheet users. `label import` checks the file against `cases.yaml` and merges it into the workspace. | Python standard library; the sheet is plain HTML and JavaScript |
162
210
  | `judge-check` | Per judge: flip rate across repeated calls (with a Wilson interval), self-agreement, majority-of-3 vote stability, accuracy and Cohen's kappa against your human labels (with intervals), position consistency and first-shown preference, and how often padding moved the verdict toward the padded answer. Verdict: `trustworthy`, `unstable`, `biased`, `misaligned`, or `not_enough_data`, with the numbers behind it. | |
163
- | `export` | promptfoo `promptfooconfig.yaml` (the full conversation as chat messages, llm-rubric asserts, and a judge that passed judge-check wired in as the grader when its rubric entry names a promptfoo `provider`), DeepEval dataset plus a `deepeval test run` file, Inspect AI dataset plus `task.py`, plain JSONL. The manifest lists file hashes and which judges passed. | |
211
+ | `export` | promptfoo `promptfooconfig.yaml` (the full conversation as chat messages, llm-rubric asserts, and a judge that passed judge-check wired in as the grader when its rubric entry names a promptfoo `provider`), DeepEval dataset plus a `deepeval test run` file (the same checked judge grades every case with its exact prompt; without one, GEval), Inspect AI dataset plus `task.py` (Inspect's default `model_graded_qa` grader), plain JSONL. The manifest lists file hashes and which judges passed. | |
164
212
  | `report` | `report.md` and `report.json`: sources with sha256, counts, redactions, selection reasons, the judge table, and the limits. | |
165
213
 
166
214
  The verdict thresholds are explicit flags with defaults: flip rate at most 20% of cases,
167
215
  position consistency at least 80%, padding helps at most 10% of cases, kappa at least
168
216
  0.4 on at least 20 human-labeled cases. A judge without human labels is never called
169
- trustworthy.
217
+ trustworthy. When at least 80% of the selected cases or of the human labels share one
218
+ outcome, `select` and `judge-check` say so with the real proportions, because a judge
219
+ that always gives that answer would look accurate on them.
170
220
 
171
221
  ## Setup for agents
172
222
 
@@ -183,14 +233,22 @@ file it edits and does nothing on a second run. The workflow the agent follows i
183
233
  [`skills/eval-builder/SKILL.md`](skills/eval-builder/SKILL.md).
184
234
 
185
235
  MCP tools: `ingest`, `select`, `draft`, `list_cases`, `update_case`, `set_rubric`,
186
- `validate`, `judge_plan`, `judge_check`, `export`, `report`, `status`. The judge runner
236
+ `validate`, `judge_plan`, `label`, `label_import`, `judge_check`, `export`, `report`,
237
+ `status`. The judge runner
187
238
  is CLI only, because it executes a command.
188
239
 
189
240
  ## What it can't do
190
241
 
191
242
  - It does not write expected behavior or human labels. The agent drafts expected
192
243
  behavior with you; labels must come from people. Without labels, judge-check can
193
- tell you a judge is unstable or biased, but not that it is right.
244
+ tell you a judge is unstable or biased, but not that it is right. `label` makes the
245
+ labeling quick, but someone still has to read each case.
246
+ - Picking labels where judges disagree makes each label more informative, but those
247
+ cases are harder than average, so accuracy measured on them leans pessimistic.
248
+ `--uncertain-share 0` picks by outcome and topic only.
249
+ - The labeling sheet is a static page, so it cannot save files: the person has to
250
+ click Export (or copy the text) and import it. Unexported progress lives only in
251
+ that browser's local storage.
194
252
  - Selection is lexical. Two requests that mean the same thing in different words can
195
253
  land in different clusters, and near-duplicate detection only catches close textual
196
254
  matches.
@@ -202,16 +260,21 @@ is CLI only, because it executes a command.
202
260
  - It does not run your app, your judges or your eval. The agent (or a script you name
203
261
  with `judge-run`) calls the judge model; the exported files run the eval in promptfoo,
204
262
  DeepEval or Inspect AI.
205
- - Only the promptfoo export wires a checked judge in as the grader, and only a
206
- pointwise judge whose prompt answers in JSON (`{"pass": ..., "reason": ...}`), because
207
- promptfoo's llm-rubric cannot parse a bare "pass". DeepEval and Inspect exports use
208
- their own default graders.
263
+ - A checked judge is wired in as the grader in promptfoo and DeepEval, and only a
264
+ pointwise one. promptfoo needs its prompt to answer in JSON (`{"pass": ...,
265
+ "reason": ...}`), because llm-rubric cannot parse a bare "pass". The DeepEval test
266
+ calls Ollama judges itself; for any other provider you fill in `call_judge`. The
267
+ Inspect AI export still uses Inspect's default `model_graded_qa` grader, not your
268
+ checked judge.
209
269
 
210
270
  ## Privacy and safety
211
271
 
212
272
  Everything runs locally. eval-builder makes no network calls and sends nothing
213
273
  anywhere. The only process it starts is the judge command you name with
214
- `judge-run --enable-judge-plugin`. Redaction is on unless you pass `--no-redact`, and
274
+ `judge-run --enable-judge-plugin`. The labeling sheet is one HTML file with a content
275
+ security policy that blocks every network request; it keeps unexported labels in the
276
+ browser's local storage on that machine. Exported test files can call a model when
277
+ you run them (the DeepEval test calls the judge it was given). Redaction is on unless you pass `--no-redact`, and
215
278
  the report lists what was replaced (counts and kinds, never the values).
216
279
 
217
280
  ## License
@@ -10,18 +10,26 @@ uvx eval-builder ingest sample_logs.jsonl # or your own logs: OpenAI, Anthro
10
10
  uvx eval-builder select -n 10 --stratify category && uvx eval-builder draft
11
11
  ```
12
12
 
13
- What you get (real output, eval-builder 0.1.1 on the sample file):
13
+ ![The quickstart running against eval-builder 0.1.1 from PyPI](docs/demo.gif)
14
+
15
+ What you get (real output of the commands above, eval-builder 0.1.1 from PyPI; the
16
+ curl and the three uvx calls took 35 s in total here; the very first uvx run also
17
+ downloads about 38 MiB, mostly numpy, scipy and scikit-learn, which took 6 s here):
14
18
 
15
19
  ```
16
20
  ingested 48 traces into evalset/traces.jsonl
17
21
  sample_logs.jsonl: format=openai records=48 traces=48 skipped=0 sha256=2d5dd2295784988b
18
22
  redactions: 0 {}
23
+ next: eval-builder select -n 30 (add --stratify <metadata keys> to cover them)
19
24
  48 traces -> 16 unique (32 exact dupes, 0 near dupes) -> selected 10 (8 failures) across 4 clusters
20
25
  q121-llama-13b: failure (negative user feedback); 69% of unique traces are failures and at least 30% of picks are reserved for them
26
+ q81-alpaca-13b: failure (negative user feedback); 69% of unique traces are failures and at least 30% of picks are reserved for them
21
27
  q102-alpaca-13b: covers category=reasoning (2 unique traces, 12%)
22
28
  q111-alpaca-13b: adds variety within cluster 1 (8 traces, 50%; triangle, response, person); least similar to cases already picked there
23
- ...
29
+ ... (6 more picks)
30
+ next: run draft to turn the selection into cases.yaml
24
31
  10 case(s) added, 10 total in evalset/cases.yaml
32
+ next: read the cases (list_cases, or cases.yaml), define criteria and judges (set_rubric, or rubric.yaml), fill expected_behavior and criteria per case and set status: ready (update_case), then run validate
25
33
  ```
26
34
 
27
35
  `evalset/cases.yaml` now holds 10 real conversations with a TODO where the expected
@@ -71,7 +79,10 @@ export and report in 32 turns and about 10 minutes. From its final answer:
71
79
  > Its verdicts will be unverified.
72
80
 
73
81
  It wrote every expected behavior itself because nobody was there to confirm them, and
74
- said so on each case. It did not invent human labels; it asked for them.
82
+ said so on each case. It did not invent human labels; it asked for them. That is where
83
+ this session and the earlier one both stopped, so 0.1.2 adds the missing step: `label`
84
+ writes a sheet the person labels in their browser, and `label import` brings the
85
+ labels back for judge-check ([Human labels, end to end](#human-labels-end-to-end)).
75
86
 
76
87
  An earlier session against 0.1.0 is where the promptfoo context bug fixed in 0.1.1 came
77
88
  from. The agent read the export and told the user: "Right now it only sends the final
@@ -131,6 +142,42 @@ The same check on the suite's own pass/fail judge (qwen2.5 7B, 24 cases, no huma
131
142
  labels) gives `unstable`: its verdict changed across 5 identical calls on 7 of 24
132
143
  cases (29%, interval 15% to 49%).
133
144
 
145
+ ### Human labels, end to end
146
+
147
+ ```sh
148
+ uvx eval-builder label # picks 24 cases, writes label_sheet.html
149
+ uvx eval-builder label import ~/Downloads/labels.jsonl # what the sheet's Export button saved
150
+ uvx eval-builder judge-check
151
+ ```
152
+
153
+ ![The labeling sheet: one case per screen, pass and fail buttons with keyboard shortcuts](docs/label-sheet.png)
154
+
155
+ Run on the 48 sample conversations with every model's answer kept as a case (47
156
+ cases), three local judges (1,128 calls, 0 errors), and 24 labels entered through the
157
+ sheet in headless Chrome. The labels were made by the developer (Claude Code reading
158
+ each case for him), not by an independent annotator, so read this as a demonstration
159
+ of the flow. Details and every file: [`examples/sample-labeling/`](examples/sample-labeling/).
160
+
161
+ - `label` split the 24 picks 12/12 between cases the judges called pass and fail; 15
162
+ are cases where the judges disagree, flip or move under padding.
163
+ - The sheet made no network requests, survived a reload mid-way, and its download was
164
+ byte-identical to the copy box. `label import` took 24 rows, rejected 0.
165
+ - `judge-check` against those labels (fail 18, pass 6):
166
+
167
+ | judge | verdict | flip rate | accuracy vs labels | kappa |
168
+ |---|---|---|---|---|
169
+ | qwen2.5:7b-instruct, temp 0 | trustworthy | 0% [0%, 8%] | 75% [55%, 88%] | 0.50 [0.15, 0.85] |
170
+ | qwen2.5:7b-instruct, temp 0.8 | trustworthy | 9% [3%, 20%] | 71% [51%, 85%] | 0.44 [0.09, 0.79] |
171
+ | llama3.2:3b | unstable | 47% [33%, 61%] | 54% [35%, 72%] | 0.12 [-0.26, 0.50] |
172
+
173
+ The qwen judges pass the default thresholds, but judge-check also warns that their
174
+ accuracy is no better than always answering "fail" (75% of the labels), and every
175
+ miss went the same way: both passed a reply that said "Here is an allegorical poem"
176
+ and then wrote no poem. With 24 labels the kappa interval runs from about 0.1 to 0.8.
177
+ The DeepEval export wired the 0.8-temperature qwen judge in with its exact prompt;
178
+ DeepEval 4.2.8 ran three logged cases through it and it made the same call on the
179
+ missing poem.
180
+
134
181
  ## How it works
135
182
 
136
183
  | step | what it does | what it uses |
@@ -140,14 +187,17 @@ cases (29%, interval 15% to 49%).
140
187
  | `draft`, `validate` | Writes `cases.yaml` and `rubric.yaml` with TODO markers. The agent fills in expected behavior and criteria with you. Validation refuses ready cases that still contain TODO or reference unknown criteria. | PyYAML |
141
188
  | `judge-plan` | Lists every judge call to make: each case N times, plus probes that swap the answer order (pairwise judges) and pad an answer with an irrelevant paragraph. | |
142
189
  | `judge-run` | Optional and off by default. Sends each request as a JSON line to a command you name (your script, your provider, your keys) and records the verdicts. eval-builder ships no API keys and no provider code. | your command |
190
+ | `label`, `label import` | Picks the ready cases a person should label (default 24): the budget is split across outcomes (the judges' consensus, or the logged failure flag before judges ran) and up to half of each share goes to cases where judges disagree, flip across repeats or move under padding. Writes `label_sheet.html`, one offline file (one case per screen, pass/fail or A/B buttons, a note, keyboard shortcuts, judge verdicts hidden, progress kept in the browser) whose Export button downloads `labels.jsonl` in the format judge-check reads, and `label_sheet.csv` for spreadsheet users. `label import` checks the file against `cases.yaml` and merges it into the workspace. | Python standard library; the sheet is plain HTML and JavaScript |
143
191
  | `judge-check` | Per judge: flip rate across repeated calls (with a Wilson interval), self-agreement, majority-of-3 vote stability, accuracy and Cohen's kappa against your human labels (with intervals), position consistency and first-shown preference, and how often padding moved the verdict toward the padded answer. Verdict: `trustworthy`, `unstable`, `biased`, `misaligned`, or `not_enough_data`, with the numbers behind it. | |
144
- | `export` | promptfoo `promptfooconfig.yaml` (the full conversation as chat messages, llm-rubric asserts, and a judge that passed judge-check wired in as the grader when its rubric entry names a promptfoo `provider`), DeepEval dataset plus a `deepeval test run` file, Inspect AI dataset plus `task.py`, plain JSONL. The manifest lists file hashes and which judges passed. | |
192
+ | `export` | promptfoo `promptfooconfig.yaml` (the full conversation as chat messages, llm-rubric asserts, and a judge that passed judge-check wired in as the grader when its rubric entry names a promptfoo `provider`), DeepEval dataset plus a `deepeval test run` file (the same checked judge grades every case with its exact prompt; without one, GEval), Inspect AI dataset plus `task.py` (Inspect's default `model_graded_qa` grader), plain JSONL. The manifest lists file hashes and which judges passed. | |
145
193
  | `report` | `report.md` and `report.json`: sources with sha256, counts, redactions, selection reasons, the judge table, and the limits. | |
146
194
 
147
195
  The verdict thresholds are explicit flags with defaults: flip rate at most 20% of cases,
148
196
  position consistency at least 80%, padding helps at most 10% of cases, kappa at least
149
197
  0.4 on at least 20 human-labeled cases. A judge without human labels is never called
150
- trustworthy.
198
+ trustworthy. When at least 80% of the selected cases or of the human labels share one
199
+ outcome, `select` and `judge-check` say so with the real proportions, because a judge
200
+ that always gives that answer would look accurate on them.
151
201
 
152
202
  ## Setup for agents
153
203
 
@@ -164,14 +214,22 @@ file it edits and does nothing on a second run. The workflow the agent follows i
164
214
  [`skills/eval-builder/SKILL.md`](skills/eval-builder/SKILL.md).
165
215
 
166
216
  MCP tools: `ingest`, `select`, `draft`, `list_cases`, `update_case`, `set_rubric`,
167
- `validate`, `judge_plan`, `judge_check`, `export`, `report`, `status`. The judge runner
217
+ `validate`, `judge_plan`, `label`, `label_import`, `judge_check`, `export`, `report`,
218
+ `status`. The judge runner
168
219
  is CLI only, because it executes a command.
169
220
 
170
221
  ## What it can't do
171
222
 
172
223
  - It does not write expected behavior or human labels. The agent drafts expected
173
224
  behavior with you; labels must come from people. Without labels, judge-check can
174
- tell you a judge is unstable or biased, but not that it is right.
225
+ tell you a judge is unstable or biased, but not that it is right. `label` makes the
226
+ labeling quick, but someone still has to read each case.
227
+ - Picking labels where judges disagree makes each label more informative, but those
228
+ cases are harder than average, so accuracy measured on them leans pessimistic.
229
+ `--uncertain-share 0` picks by outcome and topic only.
230
+ - The labeling sheet is a static page, so it cannot save files: the person has to
231
+ click Export (or copy the text) and import it. Unexported progress lives only in
232
+ that browser's local storage.
175
233
  - Selection is lexical. Two requests that mean the same thing in different words can
176
234
  land in different clusters, and near-duplicate detection only catches close textual
177
235
  matches.
@@ -183,16 +241,21 @@ is CLI only, because it executes a command.
183
241
  - It does not run your app, your judges or your eval. The agent (or a script you name
184
242
  with `judge-run`) calls the judge model; the exported files run the eval in promptfoo,
185
243
  DeepEval or Inspect AI.
186
- - Only the promptfoo export wires a checked judge in as the grader, and only a
187
- pointwise judge whose prompt answers in JSON (`{"pass": ..., "reason": ...}`), because
188
- promptfoo's llm-rubric cannot parse a bare "pass". DeepEval and Inspect exports use
189
- their own default graders.
244
+ - A checked judge is wired in as the grader in promptfoo and DeepEval, and only a
245
+ pointwise one. promptfoo needs its prompt to answer in JSON (`{"pass": ...,
246
+ "reason": ...}`), because llm-rubric cannot parse a bare "pass". The DeepEval test
247
+ calls Ollama judges itself; for any other provider you fill in `call_judge`. The
248
+ Inspect AI export still uses Inspect's default `model_graded_qa` grader, not your
249
+ checked judge.
190
250
 
191
251
  ## Privacy and safety
192
252
 
193
253
  Everything runs locally. eval-builder makes no network calls and sends nothing
194
254
  anywhere. The only process it starts is the judge command you name with
195
- `judge-run --enable-judge-plugin`. Redaction is on unless you pass `--no-redact`, and
255
+ `judge-run --enable-judge-plugin`. The labeling sheet is one HTML file with a content
256
+ security policy that blocks every network request; it keeps unexported labels in the
257
+ browser's local storage on that machine. Exported test files can call a model when
258
+ you run them (the DeepEval test calls the judge it was given). Redaction is on unless you pass `--no-redact`, and
196
259
  the report lists what was replaced (counts and kinds, never the values).
197
260
 
198
261
  ## License
Binary file
@@ -0,0 +1,25 @@
1
+ Output docs/demo.gif
2
+ Set FontSize 14
3
+ Set Width 1600
4
+ Set Height 900
5
+ Set Theme "Builtin Dark"
6
+ Set TypingSpeed 30ms
7
+ Hide
8
+ Type "cd \"$(mktemp -d)\" && clear"
9
+ Enter
10
+ Show
11
+ Type "curl -sLO https://raw.githubusercontent.com/Abelo9996/eval-builder/main/examples/sample_logs.jsonl"
12
+ Enter
13
+ Wait
14
+ Type "uvx eval-builder ingest sample_logs.jsonl"
15
+ Enter
16
+ Wait
17
+ Sleep 1.5s
18
+ Type "uvx eval-builder select -n 10 --stratify category"
19
+ Enter
20
+ Wait+Screen@20s /next: run draft/
21
+ Sleep 2s
22
+ Type "uvx eval-builder draft && grep -m1 -A10 'id: case-002' evalset/cases.yaml | cut -c1-120"
23
+ Enter
24
+ Wait+Screen@20s /role: user/
25
+ Sleep 5s
Binary file
@@ -0,0 +1,79 @@
1
+ # Labeling example
2
+
3
+ The human-labeling step, end to end, on the 48 conversations in
4
+ [`../sample_logs.jsonl`](../sample_logs.jsonl) (16 MT-Bench questions, each answered
5
+ by gpt-4, alpaca-13b and llama-13b; CC BY 4.0). Everything here was produced by
6
+ [`run.sh`](run.sh) on an Apple M4 MacBook (16 GB) on October 8 and 9, 2026, with
7
+ eval-builder 0.1.2 from this repository.
8
+
9
+ **Who labeled.** The labels in [`evalset/labels.jsonl`](evalset/labels.jsonl) were made
10
+ by the developer, not by an independent annotator: Claude Code, working for the
11
+ developer, read each picked case in full (earlier turns, reply, expected behavior) and
12
+ recorded a decision with a note for the close calls ([`decisions.json`](decisions.json)).
13
+ [`drive_sheet.py`](drive_sheet.py) then entered those decisions through the real
14
+ `label_sheet.html` in headless Chrome, using the keyboard shortcuts, and clicked Export.
15
+ The expected behavior in [`fill.py`](fill.py) was written the same way. Treat the
16
+ numbers below as a demonstration of the flow, not as ground truth about these judges.
17
+
18
+ ## Steps and what they printed
19
+
20
+ 1. `select --dedupe-on input+output --near-dup 0.99` kept each model's answer: 48
21
+ traces, 47 unique (two answers to one question were near-identical). All 47 became
22
+ ready cases with one expected behavior per question.
23
+ 2. Three local judges through Ollama (`examples/judges/ollama_judge.py`), one pass/fail
24
+ prompt that answers in JSON, 5 trials per case plus 3 with an irrelevant paragraph
25
+ appended to the reply: 1,128 calls, 0 errors (qwen2.5:7b-instruct at temperature
26
+ 0.8: 1,114 s; the same model at temperature 0: 1,130 s; llama3.2:3b: 589 s; the
27
+ machine was shared with other jobs).
28
+ 3. `judge-check` without labels: llama3.2-3b `unstable` (verdict changed across repeats
29
+ on 47% of cases), both qwen judges `not_enough_data` ("stable, but only 0 case(s)
30
+ have human labels").
31
+ 4. `label` picked 24 of the 47 cases: 12 where the judges' consensus was fail and 12
32
+ where it was pass; 15 of the 24 are cases where judges disagree, flip or move under
33
+ padding ([`evalset/label_plan.json`](evalset/label_plan.json) has every reason).
34
+ 5. The sheet in headless Chrome ([`sheet-run/drive_log.txt`](sheet-run/drive_log.txt)):
35
+ 24 cases, a reload after 13 labels kept "13 of 24 labeled", the download
36
+ (`labels.jsonl`, 4,292 bytes) was identical to the copy box, 2 requests in total,
37
+ none of them to anything but `file:`, `blob:` or `data:` URLs, no console errors. Screenshots:
38
+ [first case](sheet-run/sheet-first-case.png), [phone width](sheet-run/sheet-phone.png),
39
+ [export](sheet-run/sheet-export.png).
40
+ 6. `label import`: 24 imported, 0 rejected; labels fail 18, pass 6.
41
+ 7. `judge-check` with the labels ([`evalset/report.md`](evalset/report.md)):
42
+
43
+ | judge | verdict | flip rate | accuracy vs labels | kappa | padding helped |
44
+ |---|---|---|---|---|---|
45
+ | qwen2.5:7b-instruct, temp 0 | trustworthy | 0% [0%, 8%] | 75% [55%, 88%] | 0.50 [0.15, 0.85] | 0% [0%, 8%] |
46
+ | qwen2.5:7b-instruct, temp 0.8 | trustworthy | 9% [3%, 20%] | 71% [51%, 85%] | 0.44 [0.09, 0.79] | 0% [0%, 8%] |
47
+ | llama3.2:3b | unstable | 47% [33%, 61%] | 54% [35%, 72%] | 0.12 [-0.26, 0.50] | 4% [1%, 14%] |
48
+
49
+ n = 47 cases for flip rate and padding, 24 labeled cases for accuracy and kappa;
50
+ brackets are 95% intervals.
51
+
52
+ ## What it shows, read plainly
53
+
54
+ - Both qwen judges clear the default thresholds (kappa at least 0.4 on at least 20
55
+ labels), so the export wires `qwen2.5-7b` (the first that passed) into promptfoo and
56
+ DeepEval. The kappa intervals are wide, from about 0.1 to 0.8: 24 labels cannot pin
57
+ the agreement down.
58
+ - Their accuracy (71% and 75%) is no better than always answering "fail" (75% of the
59
+ labels are fail), and judge-check now says so in a warning. Every miss went the same
60
+ way: the judge passed a reply the labels failed. Both qwen judges passed case-018
61
+ (llama-13b answered "Here is an allegorical poem that illustrates the above:" and
62
+ no poem) and case-019 (alpaca-13b's rewrite in which most sentences do not start with
63
+ "A").
64
+ - The labels came out 18 fail to 6 pass even though `label` split the picks 12/12 by
65
+ the judges' consensus: the judges said pass far more often than the labels did.
66
+ - llama3.2:3b changes its verdict on almost half the cases between identical calls.
67
+
68
+ ## DeepEval with the checked judge
69
+
70
+ `export` wrote `evalset/exports/deepeval/judge.json` (judge `qwen2.5-7b`, its exact
71
+ prompt, provider `ollama:chat:qwen2.5:7b-instruct` at temperature 0.8). DeepEval 4.2.8
72
+ ran three of the cases on the logged outputs
73
+ (`EVAL_BUILDER_USE_OBSERVED=1 pytest test_eval_builder.py -k "case-018 or case-020 or case-040"`):
74
+ case-020 passed, case-040 failed with the judge's reason
75
+ `{"pass": false, "reason": "Incorrect circumradius calculation."}`, and case-018 (the
76
+ missing poem) passed, the same mistake judge-check found.
77
+
78
+ `evalset/judge_requests.jsonl` (4.7 MB of rendered prompts) is not committed; `judge-plan`
79
+ rebuilds it from `cases.yaml` and `rubric.yaml`.
@@ -0,0 +1,98 @@
1
+ {
2
+ "case-006": {
3
+ "label": "fail",
4
+ "note": ""
5
+ },
6
+ "case-007": {
7
+ "label": "fail",
8
+ "note": "same textbook definitions, not for a five-year-old"
9
+ },
10
+ "case-008": {
11
+ "label": "pass",
12
+ "note": ""
13
+ },
14
+ "case-009": {
15
+ "label": "fail",
16
+ "note": "repeats one definition, not simplified"
17
+ },
18
+ "case-010": {
19
+ "label": "fail",
20
+ "note": "one sentence, no program"
21
+ },
22
+ "case-012": {
23
+ "label": "fail",
24
+ "note": ""
25
+ },
26
+ "case-013": {
27
+ "label": "fail",
28
+ "note": "wrong start values and sums two terms, not three"
29
+ },
30
+ "case-015": {
31
+ "label": "fail",
32
+ "note": "Fibonacci-style, wrong recurrence"
33
+ },
34
+ "case-016": {
35
+ "label": "fail",
36
+ "note": "names the stages literally; not an allegory"
37
+ },
38
+ "case-018": {
39
+ "label": "fail",
40
+ "note": "no poem"
41
+ },
42
+ "case-019": {
43
+ "label": "fail",
44
+ "note": "most sentences do not start with A"
45
+ },
46
+ "case-023": {
47
+ "label": "pass",
48
+ "note": "names one improvement; misses that the email is not short"
49
+ },
50
+ "case-024": {
51
+ "label": "fail",
52
+ "note": "answers a different message"
53
+ },
54
+ "case-025": {
55
+ "label": "fail",
56
+ "note": "'US President' and 'Lewis' are not specific people"
57
+ },
58
+ "case-027": {
59
+ "label": "fail",
60
+ "note": "three lines, Washington instead of Roosevelt, da Vinci under science"
61
+ },
62
+ "case-029": {
63
+ "label": "pass",
64
+ "note": "explains why it is a contradiction"
65
+ },
66
+ "case-031": {
67
+ "label": "pass",
68
+ "note": "says no, without explaining why"
69
+ },
70
+ "case-032": {
71
+ "label": "pass",
72
+ "note": ""
73
+ },
74
+ "case-034": {
75
+ "label": "fail",
76
+ "note": "says $4000, should be $2000"
77
+ },
78
+ "case-036": {
79
+ "label": "fail",
80
+ "note": "ratings reordered and only one date"
81
+ },
82
+ "case-039": {
83
+ "label": "fail",
84
+ "note": "12 is wrong; 5*pi"
85
+ },
86
+ "case-042": {
87
+ "label": "fail",
88
+ "note": "restates definitions, no real assumptions examined"
89
+ },
90
+ "case-043": {
91
+ "label": "pass",
92
+ "note": ""
93
+ },
94
+ "case-047": {
95
+ "label": "fail",
96
+ "note": "repeats a wrong claim"
97
+ }
98
+ }