eval-builder 0.1.2__tar.gz → 0.1.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (122) hide show
  1. {eval_builder-0.1.2 → eval_builder-0.1.3}/CHANGELOG.md +7 -0
  2. {eval_builder-0.1.2 → eval_builder-0.1.3}/PKG-INFO +14 -12
  3. {eval_builder-0.1.2 → eval_builder-0.1.3}/README.md +13 -11
  4. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/README.md +13 -11
  5. eval_builder-0.1.3/examples/sample-labeling/evalset/exports/manifest.json +60 -0
  6. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/exports/promptfoo/promptfooconfig.yaml +3 -28
  7. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/judge_check.json +17 -19
  8. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/report.json +28 -60
  9. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/report.md +9 -10
  10. {eval_builder-0.1.2 → eval_builder-0.1.3}/pyproject.toml +1 -1
  11. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/__init__.py +1 -1
  12. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/judge/check.py +15 -2
  13. {eval_builder-0.1.2 → eval_builder-0.1.3}/tests/test_judge_check.py +23 -0
  14. {eval_builder-0.1.2 → eval_builder-0.1.3}/uv.lock +1 -1
  15. eval_builder-0.1.2/examples/sample-labeling/evalset/exports/deepeval/judge.json +0 -10
  16. eval_builder-0.1.2/examples/sample-labeling/evalset/exports/manifest.json +0 -90
  17. {eval_builder-0.1.2 → eval_builder-0.1.3}/.github/workflows/ci.yml +0 -0
  18. {eval_builder-0.1.2 → eval_builder-0.1.3}/.github/workflows/release.yml +0 -0
  19. {eval_builder-0.1.2 → eval_builder-0.1.3}/.gitignore +0 -0
  20. {eval_builder-0.1.2 → eval_builder-0.1.3}/AGENTS.md +0 -0
  21. {eval_builder-0.1.2 → eval_builder-0.1.3}/CONTRIBUTING.md +0 -0
  22. {eval_builder-0.1.2 → eval_builder-0.1.3}/LICENSE +0 -0
  23. {eval_builder-0.1.2 → eval_builder-0.1.3}/SECURITY.md +0 -0
  24. {eval_builder-0.1.2 → eval_builder-0.1.3}/docs/demo.gif +0 -0
  25. {eval_builder-0.1.2 → eval_builder-0.1.3}/docs/demo.tape +0 -0
  26. {eval_builder-0.1.2 → eval_builder-0.1.3}/docs/label-sheet.png +0 -0
  27. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/judges/control_judges.py +0 -0
  28. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/judges/ollama_judge.py +0 -0
  29. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/README.md +0 -0
  30. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/expected_behaviors.yaml +0 -0
  31. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/fill_suite.py +0 -0
  32. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/judges/cases.yaml +0 -0
  33. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/judges/judge_check.json +0 -0
  34. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/judges/judge_run_log.json +0 -0
  35. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/judges/judgments.jsonl +0 -0
  36. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/judges/labels.jsonl +0 -0
  37. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/judges/report.json +0 -0
  38. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/judges/report.md +0 -0
  39. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/judges/rubric.yaml +0 -0
  40. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/prepare.py +0 -0
  41. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/run.sh +0 -0
  42. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/suite/cases.yaml +0 -0
  43. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/suite/exports/deepeval/dataset.json +0 -0
  44. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/suite/exports/deepeval/test_eval_builder.py +0 -0
  45. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/suite/exports/inspect/dataset.jsonl +0 -0
  46. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/suite/exports/inspect/task.py +0 -0
  47. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/suite/exports/jsonl/cases.jsonl +0 -0
  48. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/suite/exports/manifest.json +0 -0
  49. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/suite/exports/promptfoo/promptfooconfig.yaml +0 -0
  50. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/suite/ingest.json +0 -0
  51. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/suite/judge_check.json +0 -0
  52. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/suite/judge_run_log.json +0 -0
  53. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/suite/judgments.jsonl +0 -0
  54. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/suite/report.json +0 -0
  55. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/suite/report.md +0 -0
  56. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/suite/rubric.yaml +0 -0
  57. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/suite/selection.json +0 -0
  58. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/mt-bench/verify_exports.sh +0 -0
  59. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/decisions.json +0 -0
  60. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/drive_sheet.py +0 -0
  61. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/cases.yaml +0 -0
  62. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/exports/deepeval/dataset.json +0 -0
  63. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/exports/deepeval/test_eval_builder.py +0 -0
  64. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/exports/inspect/dataset.jsonl +0 -0
  65. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/exports/inspect/task.py +0 -0
  66. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/exports/jsonl/cases.jsonl +0 -0
  67. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/ingest.json +0 -0
  68. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/judge_run_log.json +0 -0
  69. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/judgments.jsonl +0 -0
  70. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/label_plan.json +0 -0
  71. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/label_sheet.csv +0 -0
  72. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/label_sheet.html +0 -0
  73. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/labels.jsonl +0 -0
  74. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/rubric.yaml +0 -0
  75. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/selection.json +0 -0
  76. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/evalset/traces.jsonl +0 -0
  77. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/fill.py +0 -0
  78. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/run.sh +0 -0
  79. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/sheet-run/drive_log.txt +0 -0
  80. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/sheet-run/labels.jsonl +0 -0
  81. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/sheet-run/sheet-export.png +0 -0
  82. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/sheet-run/sheet-first-case.png +0 -0
  83. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample-labeling/sheet-run/sheet-phone.png +0 -0
  84. {eval_builder-0.1.2 → eval_builder-0.1.3}/examples/sample_logs.jsonl +0 -0
  85. {eval_builder-0.1.2 → eval_builder-0.1.3}/skills/eval-builder/SKILL.md +0 -0
  86. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/balance.py +0 -0
  87. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/cli.py +0 -0
  88. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/draft.py +0 -0
  89. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/export.py +0 -0
  90. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/ingest/__init__.py +0 -0
  91. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/ingest/formats.py +0 -0
  92. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/io.py +0 -0
  93. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/judge/__init__.py +0 -0
  94. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/judge/plan.py +0 -0
  95. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/judge/run.py +0 -0
  96. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/judge/stats.py +0 -0
  97. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/label.py +0 -0
  98. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/label_sheet.py +0 -0
  99. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/mcp_server.py +0 -0
  100. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/redact.py +0 -0
  101. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/report.py +0 -0
  102. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/schema.py +0 -0
  103. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/select.py +0 -0
  104. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/setup_agents.py +0 -0
  105. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/status.py +0 -0
  106. {eval_builder-0.1.2 → eval_builder-0.1.3}/src/eval_builder/workspace.py +0 -0
  107. {eval_builder-0.1.2 → eval_builder-0.1.3}/tests/conftest.py +0 -0
  108. {eval_builder-0.1.2 → eval_builder-0.1.3}/tests/fixtures/anthropic_messages.json +0 -0
  109. {eval_builder-0.1.2 → eval_builder-0.1.3}/tests/fixtures/generic.jsonl +0 -0
  110. {eval_builder-0.1.2 → eval_builder-0.1.3}/tests/fixtures/langfuse_export.json +0 -0
  111. {eval_builder-0.1.2 → eval_builder-0.1.3}/tests/fixtures/openai_chat.jsonl +0 -0
  112. {eval_builder-0.1.2 → eval_builder-0.1.3}/tests/fixtures/otel_genai.json +0 -0
  113. {eval_builder-0.1.2 → eval_builder-0.1.3}/tests/test_cli_mcp.py +0 -0
  114. {eval_builder-0.1.2 → eval_builder-0.1.3}/tests/test_draft.py +0 -0
  115. {eval_builder-0.1.2 → eval_builder-0.1.3}/tests/test_export.py +0 -0
  116. {eval_builder-0.1.2 → eval_builder-0.1.3}/tests/test_ingest.py +0 -0
  117. {eval_builder-0.1.2 → eval_builder-0.1.3}/tests/test_judge_run.py +0 -0
  118. {eval_builder-0.1.2 → eval_builder-0.1.3}/tests/test_judge_stats.py +0 -0
  119. {eval_builder-0.1.2 → eval_builder-0.1.3}/tests/test_label.py +0 -0
  120. {eval_builder-0.1.2 → eval_builder-0.1.3}/tests/test_report.py +0 -0
  121. {eval_builder-0.1.2 → eval_builder-0.1.3}/tests/test_select.py +0 -0
  122. {eval_builder-0.1.2 → eval_builder-0.1.3}/tests/test_setup.py +0 -0
@@ -1,5 +1,12 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.1.3
4
+
5
+ ### Fixed
6
+
7
+ - `judge-check` no longer calls a judge `trustworthy` when its accuracy against the human labels is no better than always giving the most common label. Kappa alone let a lenient judge pass: on the labeling example (18 of 24 labels "fail") both qwen2.5 judges cleared kappa 0.4 at 71% and 75% accuracy, the same as always answering "fail". They are now `misaligned`, with the baseline in the reason. Set `Thresholds(beat_majority_baseline=False)` to get the old behavior.
8
+ - Because no judge passes on that example any more, `export` no longer wires one into promptfoo or DeepEval there; the example README says what changed.
9
+
3
10
  ## 0.1.2 (2026-10-08)
4
11
 
5
12
  Human labels. No judge can be called trustworthy without them, and both real agent
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: eval-builder
3
- Version: 0.1.2
3
+ Version: 0.1.3
4
4
  Summary: Turn real LLM app logs into an eval suite and measure which LLM judges you can trust.
5
5
  Author: Abel Yagubyan
6
6
  License-Expression: MIT
@@ -173,8 +173,8 @@ uvx eval-builder judge-check
173
173
 
174
174
  Run on the 48 sample conversations with every model's answer kept as a case (47
175
175
  cases), three local judges (1,128 calls, 0 errors), and 24 labels entered through the
176
- sheet in headless Chrome. The labels were made by the developer (Claude Code reading
177
- each case for him), not by an independent annotator, so read this as a demonstration
176
+ sheet in headless Chrome. The labels were made on the developer's side (Claude Code
177
+ reading each case), not by an independent annotator, so read this as a demonstration
178
178
  of the flow. Details and every file: [`examples/sample-labeling/`](examples/sample-labeling/).
179
179
 
180
180
  - `label` split the 24 picks 12/12 between cases the judges called pass and fail; 15
@@ -185,17 +185,19 @@ of the flow. Details and every file: [`examples/sample-labeling/`](examples/samp
185
185
 
186
186
  | judge | verdict | flip rate | accuracy vs labels | kappa |
187
187
  |---|---|---|---|---|
188
- | qwen2.5:7b-instruct, temp 0 | trustworthy | 0% [0%, 8%] | 75% [55%, 88%] | 0.50 [0.15, 0.85] |
189
- | qwen2.5:7b-instruct, temp 0.8 | trustworthy | 9% [3%, 20%] | 71% [51%, 85%] | 0.44 [0.09, 0.79] |
188
+ | qwen2.5:7b-instruct, temp 0 | misaligned | 0% [0%, 8%] | 75% [55%, 88%] | 0.50 [0.15, 0.85] |
189
+ | qwen2.5:7b-instruct, temp 0.8 | misaligned | 9% [3%, 20%] | 71% [51%, 85%] | 0.44 [0.09, 0.79] |
190
190
  | llama3.2:3b | unstable | 47% [33%, 61%] | 54% [35%, 72%] | 0.12 [-0.26, 0.50] |
191
191
 
192
- The qwen judges pass the default thresholds, but judge-check also warns that their
193
- accuracy is no better than always answering "fail" (75% of the labels), and every
194
- miss went the same way: both passed a reply that said "Here is an allegorical poem"
195
- and then wrote no poem. With 24 labels the kappa interval runs from about 0.1 to 0.8.
196
- The DeepEval export wired the 0.8-temperature qwen judge in with its exact prompt;
197
- DeepEval 4.2.8 ran three logged cases through it and it made the same call on the
198
- missing poem.
192
+ The qwen judges clear the kappa threshold (0.4), but their accuracy is no better than
193
+ always answering "fail" (75% of the labels), so since 0.1.3 judge-check calls them
194
+ `misaligned` instead of trustworthy (0.1.2 called them trustworthy; that was the bug).
195
+ Every miss went the same way: both passed a reply that said "Here is an allegorical
196
+ poem" and then wrote no poem. With 24 labels the kappa interval runs from about 0.1
197
+ to 0.8. Under 0.1.2 the DeepEval export wired the 0.8-temperature qwen judge in with
198
+ its exact prompt; DeepEval 4.2.8 ran three logged cases through it and it made the
199
+ same call on the missing poem. Now no judge is wired in unless you force one with
200
+ `export --judge <id>`, which warns.
199
201
 
200
202
  ## How it works
201
203
 
@@ -154,8 +154,8 @@ uvx eval-builder judge-check
154
154
 
155
155
  Run on the 48 sample conversations with every model's answer kept as a case (47
156
156
  cases), three local judges (1,128 calls, 0 errors), and 24 labels entered through the
157
- sheet in headless Chrome. The labels were made by the developer (Claude Code reading
158
- each case for him), not by an independent annotator, so read this as a demonstration
157
+ sheet in headless Chrome. The labels were made on the developer's side (Claude Code
158
+ reading each case), not by an independent annotator, so read this as a demonstration
159
159
  of the flow. Details and every file: [`examples/sample-labeling/`](examples/sample-labeling/).
160
160
 
161
161
  - `label` split the 24 picks 12/12 between cases the judges called pass and fail; 15
@@ -166,17 +166,19 @@ of the flow. Details and every file: [`examples/sample-labeling/`](examples/samp
166
166
 
167
167
  | judge | verdict | flip rate | accuracy vs labels | kappa |
168
168
  |---|---|---|---|---|
169
- | qwen2.5:7b-instruct, temp 0 | trustworthy | 0% [0%, 8%] | 75% [55%, 88%] | 0.50 [0.15, 0.85] |
170
- | qwen2.5:7b-instruct, temp 0.8 | trustworthy | 9% [3%, 20%] | 71% [51%, 85%] | 0.44 [0.09, 0.79] |
169
+ | qwen2.5:7b-instruct, temp 0 | misaligned | 0% [0%, 8%] | 75% [55%, 88%] | 0.50 [0.15, 0.85] |
170
+ | qwen2.5:7b-instruct, temp 0.8 | misaligned | 9% [3%, 20%] | 71% [51%, 85%] | 0.44 [0.09, 0.79] |
171
171
  | llama3.2:3b | unstable | 47% [33%, 61%] | 54% [35%, 72%] | 0.12 [-0.26, 0.50] |
172
172
 
173
- The qwen judges pass the default thresholds, but judge-check also warns that their
174
- accuracy is no better than always answering "fail" (75% of the labels), and every
175
- miss went the same way: both passed a reply that said "Here is an allegorical poem"
176
- and then wrote no poem. With 24 labels the kappa interval runs from about 0.1 to 0.8.
177
- The DeepEval export wired the 0.8-temperature qwen judge in with its exact prompt;
178
- DeepEval 4.2.8 ran three logged cases through it and it made the same call on the
179
- missing poem.
173
+ The qwen judges clear the kappa threshold (0.4), but their accuracy is no better than
174
+ always answering "fail" (75% of the labels), so since 0.1.3 judge-check calls them
175
+ `misaligned` instead of trustworthy (0.1.2 called them trustworthy; that was the bug).
176
+ Every miss went the same way: both passed a reply that said "Here is an allegorical
177
+ poem" and then wrote no poem. With 24 labels the kappa interval runs from about 0.1
178
+ to 0.8. Under 0.1.2 the DeepEval export wired the 0.8-temperature qwen judge in with
179
+ its exact prompt; DeepEval 4.2.8 ran three logged cases through it and it made the
180
+ same call on the missing poem. Now no judge is wired in unless you force one with
181
+ `export --judge <id>`, which warns.
180
182
 
181
183
  ## How it works
182
184
 
@@ -42,8 +42,8 @@ numbers below as a demonstration of the flow, not as ground truth about these ju
42
42
 
43
43
  | judge | verdict | flip rate | accuracy vs labels | kappa | padding helped |
44
44
  |---|---|---|---|---|---|
45
- | qwen2.5:7b-instruct, temp 0 | trustworthy | 0% [0%, 8%] | 75% [55%, 88%] | 0.50 [0.15, 0.85] | 0% [0%, 8%] |
46
- | qwen2.5:7b-instruct, temp 0.8 | trustworthy | 9% [3%, 20%] | 71% [51%, 85%] | 0.44 [0.09, 0.79] | 0% [0%, 8%] |
45
+ | qwen2.5:7b-instruct, temp 0 | misaligned | 0% [0%, 8%] | 75% [55%, 88%] | 0.50 [0.15, 0.85] | 0% [0%, 8%] |
46
+ | qwen2.5:7b-instruct, temp 0.8 | misaligned | 9% [3%, 20%] | 71% [51%, 85%] | 0.44 [0.09, 0.79] | 0% [0%, 8%] |
47
47
  | llama3.2:3b | unstable | 47% [33%, 61%] | 54% [35%, 72%] | 0.12 [-0.26, 0.50] | 4% [1%, 14%] |
48
48
 
49
49
  n = 47 cases for flip rate and padding, 24 labeled cases for accuracy and kappa;
@@ -51,12 +51,12 @@ brackets are 95% intervals.
51
51
 
52
52
  ## What it shows, read plainly
53
53
 
54
- - Both qwen judges clear the default thresholds (kappa at least 0.4 on at least 20
55
- labels), so the export wires `qwen2.5-7b` (the first that passed) into promptfoo and
56
- DeepEval. The kappa intervals are wide, from about 0.1 to 0.8: 24 labels cannot pin
57
- the agreement down.
58
- - Their accuracy (71% and 75%) is no better than always answering "fail" (75% of the
59
- labels are fail), and judge-check now says so in a warning. Every miss went the same
54
+ - Both qwen judges clear the kappa threshold (at least 0.4 on at least 20 labels), but
55
+ their accuracy (71% and 75%) is no better than always answering "fail" (75% of the
56
+ labels are fail). eval-builder 0.1.2 still called them trustworthy and wired
57
+ `qwen2.5-7b` into promptfoo and DeepEval; 0.1.3 requires a judge to beat that
58
+ baseline, so both are now `misaligned` and no judge is wired in. The kappa intervals
59
+ are wide, from about 0.1 to 0.8: 24 labels cannot pin the agreement down. Every miss went the same
60
60
  way: the judge passed a reply the labels failed. Both qwen judges passed case-018
61
61
  (llama-13b answered "Here is an allegorical poem that illustrates the above:" and
62
62
  no poem) and case-019 (alpaca-13b's rewrite in which most sentences do not start with
@@ -65,15 +65,17 @@ brackets are 95% intervals.
65
65
  the judges' consensus: the judges said pass far more often than the labels did.
66
66
  - llama3.2:3b changes its verdict on almost half the cases between identical calls.
67
67
 
68
- ## DeepEval with the checked judge
68
+ ## DeepEval with the checked judge (run under 0.1.2)
69
69
 
70
- `export` wrote `evalset/exports/deepeval/judge.json` (judge `qwen2.5-7b`, its exact
70
+ Under 0.1.2, `export` wrote `evalset/exports/deepeval/judge.json` (removed by the 0.1.3 re-export) (judge `qwen2.5-7b`, its exact
71
71
  prompt, provider `ollama:chat:qwen2.5:7b-instruct` at temperature 0.8). DeepEval 4.2.8
72
72
  ran three of the cases on the logged outputs
73
73
  (`EVAL_BUILDER_USE_OBSERVED=1 pytest test_eval_builder.py -k "case-018 or case-020 or case-040"`):
74
74
  case-020 passed, case-040 failed with the judge's reason
75
75
  `{"pass": false, "reason": "Incorrect circumradius calculation."}`, and case-018 (the
76
- missing poem) passed, the same mistake judge-check found.
76
+ missing poem) passed, the same mistake judge-check found. With 0.1.3 the same export
77
+ falls back to GEval unless you pass `export --judge qwen2.5-7b`, which warns that the
78
+ judge did not pass.
77
79
 
78
80
  `evalset/judge_requests.jsonl` (4.7 MB of rendered prompts) is not committed; `judge-plan`
79
81
  rebuilds it from `cases.yaml` and `rubric.yaml`.
@@ -0,0 +1,60 @@
1
+ {
2
+ "cases": 47,
3
+ "formats": [
4
+ "promptfoo",
5
+ "deepeval",
6
+ "inspect",
7
+ "jsonl"
8
+ ],
9
+ "files": [
10
+ {
11
+ "path": "exports/promptfoo/promptfooconfig.yaml",
12
+ "sha256": "07a874ff9a93431609865407d7b5759e39fbe50d9eb5a993f6ba8e3a8b852cbd"
13
+ },
14
+ {
15
+ "path": "exports/deepeval/dataset.json",
16
+ "sha256": "f6cf982eeef58e135e8b20bd9bd2a39f6965e1c2bc8e8ceb39e38f07b8c17536"
17
+ },
18
+ {
19
+ "path": "exports/deepeval/test_eval_builder.py",
20
+ "sha256": "05f445309c2983afd8a9cfb675dc7fd2a97f32f35ea2d918442a7bb391c270f7"
21
+ },
22
+ {
23
+ "path": "exports/inspect/dataset.jsonl",
24
+ "sha256": "63fb2547e1e810076647ebbd7c8a23cadd290ed8ad28fd1fc886b5b17b2b9215"
25
+ },
26
+ {
27
+ "path": "exports/inspect/task.py",
28
+ "sha256": "df284b6b6dedeb833e283cd9226241c17a84bed70924d097e9a6c45cba69c36e"
29
+ },
30
+ {
31
+ "path": "exports/jsonl/cases.jsonl",
32
+ "sha256": "d10eff47dcc67a113871b634bdbf8a705ce71ab6d485f0b109a06934d9c2bdef"
33
+ }
34
+ ],
35
+ "judge_check": {
36
+ "trustworthy": [],
37
+ "verdicts": {
38
+ "llama3.2-3b": "unstable",
39
+ "qwen2.5-7b": "misaligned",
40
+ "qwen2.5-7b-temp0": "misaligned"
41
+ },
42
+ "thresholds": {
43
+ "min_trials": 3,
44
+ "min_cases": 10,
45
+ "max_flip_rate": 0.2,
46
+ "min_position_consistency": 0.8,
47
+ "max_toward_padded_rate": 0.1,
48
+ "min_kappa": 0.4,
49
+ "min_labeled": 20,
50
+ "beat_majority_baseline": true
51
+ }
52
+ },
53
+ "trusted_judge_prompts": {},
54
+ "promptfoo_grader": null,
55
+ "deepeval_grader": null,
56
+ "notes": [
57
+ "DeepEval: no judge passed judge-check, so test_eval_builder.py grades with GEval and the judge model configured in DeepEval (not a judge checked here)",
58
+ "Inspect AI: task.py grades with Inspect's default model_graded_qa scorer, not a judge checked here"
59
+ ]
60
+ }
@@ -4,39 +4,14 @@
4
4
  # The prompt is the whole conversation as a JSON list of chat messages (vars.messages), which
5
5
  # chat providers such as openai:chat or ollama:chat accept as is. If your provider takes plain
6
6
  # text, use {{input}} (the latest user message) and {{context}} (the earlier turns) instead.
7
- # Grader: judge qwen2.5-7b from rubric.yaml (judge-check verdict: trustworthy),
8
- # with its exact prompt and model. promptfoo's llm-rubric needs the grader to answer in JSON,
9
- # {"pass": true|false, "reason": "..."}; a judge that answers with a bare word fails with
10
- # "Could not extract JSON".
7
+ # No grader is wired in: llm-rubric uses promptfoo's default grader, which
8
+ # sees only the output and the rubric. To grade with a judge you checked, give it a `provider`
9
+ # in rubric.yaml (a promptfoo provider id) and export again.
11
10
  description: MT-Bench sample, labeling example
12
11
  prompts:
13
12
  - '{{ messages | dump }}'
14
13
  providers:
15
14
  - echo
16
- defaultTest:
17
- options:
18
- provider: ollama:chat:qwen2.5:7b-instruct
19
- rubricPrompt: |-
20
- {% raw %}You are grading one reply from an AI assistant.
21
-
22
- Earlier conversation:
23
- {% endraw %}{{ context }}{% raw %}
24
-
25
- Latest user message:
26
- {% endraw %}{{ input }}{% raw %}
27
-
28
- Assistant reply to grade:
29
- {% endraw %}{{ output }}{% raw %}
30
-
31
- What a good reply does:
32
- {% endraw %}{{ expected_behavior }}{% raw %}
33
-
34
- Criteria:
35
- {% endraw %}{{ criteria }}{% raw %}
36
-
37
- Does the reply do what a good reply does and meet the criteria? Answer with JSON only:
38
- {"pass": true or false, "reason": "<one short sentence>"}
39
- {% endraw %}
40
15
  tests:
41
16
  - description: 'case-001: How do you like dancing? Can you teach me?'
42
17
  vars:
@@ -8,9 +8,9 @@
8
8
  "pass": 6
9
9
  },
10
10
  "warnings": [
11
- "judge llama3.2-3b's accuracy (54%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said pass 13 of 24 (54%), fail 11 (46%)); kappa 0.12 [-0.26, 0.50] is the number that corrects for this. Look at the cases it got wrong before relying on it",
12
- "judge qwen2.5-7b's accuracy (71%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said pass 13 of 24 (54%), fail 11 (46%)); kappa 0.44 [0.09, 0.79] is the number that corrects for this. Look at the cases it got wrong before relying on it",
13
- "judge qwen2.5-7b-temp0's accuracy (75%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said fail 12 of 24 (50%), pass 12 (50%)); kappa 0.50 [0.15, 0.85] is the number that corrects for this. Look at the cases it got wrong before relying on it"
11
+ "judge llama3.2-3b's accuracy (54%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said pass 13 of 24 (54%), fail 11 (46%); kappa 0.12 [-0.26, 0.50]). A judge has to beat that baseline to be called trustworthy; look at the cases it got wrong",
12
+ "judge qwen2.5-7b's accuracy (71%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said pass 13 of 24 (54%), fail 11 (46%); kappa 0.44 [0.09, 0.79]). A judge has to beat that baseline to be called trustworthy; look at the cases it got wrong",
13
+ "judge qwen2.5-7b-temp0's accuracy (75%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said fail 12 of 24 (50%), pass 12 (50%); kappa 0.50 [0.15, 0.85]). A judge has to beat that baseline to be called trustworthy; look at the cases it got wrong"
14
14
  ],
15
15
  "thresholds": {
16
16
  "min_trials": 3,
@@ -19,7 +19,8 @@
19
19
  "min_position_consistency": 0.8,
20
20
  "max_toward_padded_rate": 0.1,
21
21
  "min_kappa": 0.4,
22
- "min_labeled": 20
22
+ "min_labeled": 20,
23
+ "beat_majority_baseline": true
23
24
  },
24
25
  "summary": [
25
26
  {
@@ -61,7 +62,7 @@
61
62
  {
62
63
  "judge": "qwen2.5-7b",
63
64
  "mode": "pointwise",
64
- "verdict": "trustworthy",
65
+ "verdict": "misaligned",
65
66
  "cases": 47,
66
67
  "calls": 376,
67
68
  "invalid_outputs": 0,
@@ -90,13 +91,13 @@
90
91
  0.0756
91
92
  ],
92
93
  "reasons": [
93
- "stable (flip rate 9%), kappa 0.44 with humans"
94
+ "accuracy 71% is no better than always answering 'fail' (75% on these labels), so kappa 0.44 alone doesn't show the judge adds anything"
94
95
  ]
95
96
  },
96
97
  {
97
98
  "judge": "qwen2.5-7b-temp0",
98
99
  "mode": "pointwise",
99
- "verdict": "trustworthy",
100
+ "verdict": "misaligned",
100
101
  "cases": 47,
101
102
  "calls": 376,
102
103
  "invalid_outputs": 0,
@@ -125,7 +126,7 @@
125
126
  0.0756
126
127
  ],
127
128
  "reasons": [
128
- "stable (flip rate 0%), kappa 0.50 with humans"
129
+ "accuracy 75% is no better than always answering 'fail' (75% on these labels), so kappa 0.50 alone doesn't show the judge adds anything"
129
130
  ]
130
131
  }
131
132
  ],
@@ -829,15 +830,15 @@
829
830
  },
830
831
  "qwen2.5-7b": {
831
832
  "mode": "pointwise",
832
- "verdict": "trustworthy",
833
+ "verdict": "misaligned",
833
834
  "reasons": [
834
- "stable (flip rate 9%), kappa 0.44 with humans"
835
+ "accuracy 71% is no better than always answering 'fail' (75% on these labels), so kappa 0.44 alone doesn't show the judge adds anything"
835
836
  ],
836
837
  "checks": {
837
838
  "stability": "pass",
838
839
  "position": "not run",
839
840
  "verbosity": "pass",
840
- "human_agreement": "pass"
841
+ "human_agreement": "fail"
841
842
  },
842
843
  "stability": {
843
844
  "cases": 47,
@@ -1507,15 +1508,15 @@
1507
1508
  },
1508
1509
  "qwen2.5-7b-temp0": {
1509
1510
  "mode": "pointwise",
1510
- "verdict": "trustworthy",
1511
+ "verdict": "misaligned",
1511
1512
  "reasons": [
1512
- "stable (flip rate 0%), kappa 0.50 with humans"
1513
+ "accuracy 75% is no better than always answering 'fail' (75% on these labels), so kappa 0.50 alone doesn't show the judge adds anything"
1513
1514
  ],
1514
1515
  "checks": {
1515
1516
  "stability": "pass",
1516
1517
  "position": "not run",
1517
1518
  "verbosity": "pass",
1518
- "human_agreement": "pass"
1519
+ "human_agreement": "fail"
1519
1520
  },
1520
1521
  "stability": {
1521
1522
  "cases": 47,
@@ -2180,9 +2181,6 @@
2180
2181
  ]
2181
2182
  }
2182
2183
  },
2183
- "trustworthy": [
2184
- "qwen2.5-7b",
2185
- "qwen2.5-7b-temp0"
2186
- ],
2187
- "next": "export: judges ['qwen2.5-7b', 'qwen2.5-7b-temp0'] passed; export wires a passing pointwise judge into promptfoo when its rubric.yaml entry has a `provider`"
2184
+ "trustworthy": [],
2185
+ "next": "no judge passed; read each judge's reasons. Typical fixes: majority vote over 3 calls, run both answer orders and keep agreements, tighten the rubric wording, or try a stronger judge model, then run judge_plan and judge_check again"
2188
2186
  }
@@ -1,5 +1,5 @@
1
1
  {
2
- "generated_at": "2026-10-09T08:21:43+00:00",
2
+ "generated_at": "2026-10-09T08:26:01+00:00",
3
3
  "eval_builder_version": "0.1.2",
4
4
  "workspace": "examples/sample-labeling/evalset",
5
5
  "ingest": {
@@ -1280,9 +1280,9 @@
1280
1280
  "pass": 6
1281
1281
  },
1282
1282
  "warnings": [
1283
- "judge llama3.2-3b's accuracy (54%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said pass 13 of 24 (54%), fail 11 (46%)); kappa 0.12 [-0.26, 0.50] is the number that corrects for this. Look at the cases it got wrong before relying on it",
1284
- "judge qwen2.5-7b's accuracy (71%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said pass 13 of 24 (54%), fail 11 (46%)); kappa 0.44 [0.09, 0.79] is the number that corrects for this. Look at the cases it got wrong before relying on it",
1285
- "judge qwen2.5-7b-temp0's accuracy (75%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said fail 12 of 24 (50%), pass 12 (50%)); kappa 0.50 [0.15, 0.85] is the number that corrects for this. Look at the cases it got wrong before relying on it"
1283
+ "judge llama3.2-3b's accuracy (54%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said pass 13 of 24 (54%), fail 11 (46%); kappa 0.12 [-0.26, 0.50]). A judge has to beat that baseline to be called trustworthy; look at the cases it got wrong",
1284
+ "judge qwen2.5-7b's accuracy (71%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said pass 13 of 24 (54%), fail 11 (46%); kappa 0.44 [0.09, 0.79]). A judge has to beat that baseline to be called trustworthy; look at the cases it got wrong",
1285
+ "judge qwen2.5-7b-temp0's accuracy (75%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said fail 12 of 24 (50%), pass 12 (50%); kappa 0.50 [0.15, 0.85]). A judge has to beat that baseline to be called trustworthy; look at the cases it got wrong"
1286
1286
  ],
1287
1287
  "thresholds": {
1288
1288
  "min_trials": 3,
@@ -1291,7 +1291,8 @@
1291
1291
  "min_position_consistency": 0.8,
1292
1292
  "max_toward_padded_rate": 0.1,
1293
1293
  "min_kappa": 0.4,
1294
- "min_labeled": 20
1294
+ "min_labeled": 20,
1295
+ "beat_majority_baseline": true
1295
1296
  },
1296
1297
  "summary": [
1297
1298
  {
@@ -1333,7 +1334,7 @@
1333
1334
  {
1334
1335
  "judge": "qwen2.5-7b",
1335
1336
  "mode": "pointwise",
1336
- "verdict": "trustworthy",
1337
+ "verdict": "misaligned",
1337
1338
  "cases": 47,
1338
1339
  "calls": 376,
1339
1340
  "invalid_outputs": 0,
@@ -1362,13 +1363,13 @@
1362
1363
  0.0756
1363
1364
  ],
1364
1365
  "reasons": [
1365
- "stable (flip rate 9%), kappa 0.44 with humans"
1366
+ "accuracy 71% is no better than always answering 'fail' (75% on these labels), so kappa 0.44 alone doesn't show the judge adds anything"
1366
1367
  ]
1367
1368
  },
1368
1369
  {
1369
1370
  "judge": "qwen2.5-7b-temp0",
1370
1371
  "mode": "pointwise",
1371
- "verdict": "trustworthy",
1372
+ "verdict": "misaligned",
1372
1373
  "cases": 47,
1373
1374
  "calls": 376,
1374
1375
  "invalid_outputs": 0,
@@ -1397,7 +1398,7 @@
1397
1398
  0.0756
1398
1399
  ],
1399
1400
  "reasons": [
1400
- "stable (flip rate 0%), kappa 0.50 with humans"
1401
+ "accuracy 75% is no better than always answering 'fail' (75% on these labels), so kappa 0.50 alone doesn't show the judge adds anything"
1401
1402
  ]
1402
1403
  }
1403
1404
  ],
@@ -2101,15 +2102,15 @@
2101
2102
  },
2102
2103
  "qwen2.5-7b": {
2103
2104
  "mode": "pointwise",
2104
- "verdict": "trustworthy",
2105
+ "verdict": "misaligned",
2105
2106
  "reasons": [
2106
- "stable (flip rate 9%), kappa 0.44 with humans"
2107
+ "accuracy 71% is no better than always answering 'fail' (75% on these labels), so kappa 0.44 alone doesn't show the judge adds anything"
2107
2108
  ],
2108
2109
  "checks": {
2109
2110
  "stability": "pass",
2110
2111
  "position": "not run",
2111
2112
  "verbosity": "pass",
2112
- "human_agreement": "pass"
2113
+ "human_agreement": "fail"
2113
2114
  },
2114
2115
  "stability": {
2115
2116
  "cases": 47,
@@ -2779,15 +2780,15 @@
2779
2780
  },
2780
2781
  "qwen2.5-7b-temp0": {
2781
2782
  "mode": "pointwise",
2782
- "verdict": "trustworthy",
2783
+ "verdict": "misaligned",
2783
2784
  "reasons": [
2784
- "stable (flip rate 0%), kappa 0.50 with humans"
2785
+ "accuracy 75% is no better than always answering 'fail' (75% on these labels), so kappa 0.50 alone doesn't show the judge adds anything"
2785
2786
  ],
2786
2787
  "checks": {
2787
2788
  "stability": "pass",
2788
2789
  "position": "not run",
2789
2790
  "verbosity": "pass",
2790
- "human_agreement": "pass"
2791
+ "human_agreement": "fail"
2791
2792
  },
2792
2793
  "stability": {
2793
2794
  "cases": 47,
@@ -3452,11 +3453,8 @@
3452
3453
  ]
3453
3454
  }
3454
3455
  },
3455
- "trustworthy": [
3456
- "qwen2.5-7b",
3457
- "qwen2.5-7b-temp0"
3458
- ],
3459
- "next": "export: judges ['qwen2.5-7b', 'qwen2.5-7b-temp0'] passed; export wires a passing pointwise judge into promptfoo when its rubric.yaml entry has a `provider`"
3456
+ "trustworthy": [],
3457
+ "next": "no judge passed; read each judge's reasons. Typical fixes: majority vote over 3 calls, run both answer orders and keep agreements, tighten the rubric wording, or try a stronger judge model, then run judge_plan and judge_check again"
3460
3458
  },
3461
3459
  "exports": {
3462
3460
  "cases": 47,
@@ -3469,7 +3467,7 @@
3469
3467
  "files": [
3470
3468
  {
3471
3469
  "path": "exports/promptfoo/promptfooconfig.yaml",
3472
- "sha256": "23c43efb656ee17cf190fffd1100eb18986917593f6854dcdf36973ae4a6b6f3"
3470
+ "sha256": "07a874ff9a93431609865407d7b5759e39fbe50d9eb5a993f6ba8e3a8b852cbd"
3473
3471
  },
3474
3472
  {
3475
3473
  "path": "exports/deepeval/dataset.json",
@@ -3479,10 +3477,6 @@
3479
3477
  "path": "exports/deepeval/test_eval_builder.py",
3480
3478
  "sha256": "05f445309c2983afd8a9cfb675dc7fd2a97f32f35ea2d918442a7bb391c270f7"
3481
3479
  },
3482
- {
3483
- "path": "exports/deepeval/judge.json",
3484
- "sha256": "dd632220790c10ab8c788b1407cc08c0db17a28b58d4911b373b32b32bcf2fdd"
3485
- },
3486
3480
  {
3487
3481
  "path": "exports/inspect/dataset.jsonl",
3488
3482
  "sha256": "63fb2547e1e810076647ebbd7c8a23cadd290ed8ad28fd1fc886b5b17b2b9215"
@@ -3497,14 +3491,11 @@
3497
3491
  }
3498
3492
  ],
3499
3493
  "judge_check": {
3500
- "trustworthy": [
3501
- "qwen2.5-7b",
3502
- "qwen2.5-7b-temp0"
3503
- ],
3494
+ "trustworthy": [],
3504
3495
  "verdicts": {
3505
3496
  "llama3.2-3b": "unstable",
3506
- "qwen2.5-7b": "trustworthy",
3507
- "qwen2.5-7b-temp0": "trustworthy"
3497
+ "qwen2.5-7b": "misaligned",
3498
+ "qwen2.5-7b-temp0": "misaligned"
3508
3499
  },
3509
3500
  "thresholds": {
3510
3501
  "min_trials": 3,
@@ -3513,38 +3504,15 @@
3513
3504
  "min_position_consistency": 0.8,
3514
3505
  "max_toward_padded_rate": 0.1,
3515
3506
  "min_kappa": 0.4,
3516
- "min_labeled": 20
3517
- }
3518
- },
3519
- "trusted_judge_prompts": {
3520
- "qwen2.5-7b": {
3521
- "mode": "pointwise",
3522
- "labels": [
3523
- "pass",
3524
- "fail"
3525
- ],
3526
- "prompt": "You are grading one reply from an AI assistant.\n\nEarlier conversation:\n{context}\n\nLatest user message:\n{input}\n\nAssistant reply to grade:\n{output}\n\nWhat a good reply does:\n{expected_behavior}\n\nCriteria:\n{criteria}\n\nDoes the reply do what a good reply does and meet the criteria? Answer with JSON only:\n{\"pass\": true or false, \"reason\": \"<one short sentence>\"}\n"
3527
- },
3528
- "qwen2.5-7b-temp0": {
3529
- "mode": "pointwise",
3530
- "labels": [
3531
- "pass",
3532
- "fail"
3533
- ],
3534
- "prompt": "You are grading one reply from an AI assistant.\n\nEarlier conversation:\n{context}\n\nLatest user message:\n{input}\n\nAssistant reply to grade:\n{output}\n\nWhat a good reply does:\n{expected_behavior}\n\nCriteria:\n{criteria}\n\nDoes the reply do what a good reply does and meet the criteria? Answer with JSON only:\n{\"pass\": true or false, \"reason\": \"<one short sentence>\"}\n"
3507
+ "min_labeled": 20,
3508
+ "beat_majority_baseline": true
3535
3509
  }
3536
3510
  },
3537
- "promptfoo_grader": {
3538
- "judge": "qwen2.5-7b",
3539
- "provider": "ollama:chat:qwen2.5:7b-instruct",
3540
- "verdict": "trustworthy"
3541
- },
3542
- "deepeval_grader": {
3543
- "judge": "qwen2.5-7b",
3544
- "provider": "ollama:chat:qwen2.5:7b-instruct",
3545
- "verdict": "trustworthy"
3546
- },
3511
+ "trusted_judge_prompts": {},
3512
+ "promptfoo_grader": null,
3513
+ "deepeval_grader": null,
3547
3514
  "notes": [
3515
+ "DeepEval: no judge passed judge-check, so test_eval_builder.py grades with GEval and the judge model configured in DeepEval (not a judge checked here)",
3548
3516
  "Inspect AI: task.py grades with Inspect's default model_graded_qa scorer, not a judge checked here"
3549
3517
  ]
3550
3518
  },
@@ -1,6 +1,6 @@
1
1
  # Labeling example on the 48-conversation sample
2
2
 
3
- Generated 2026-10-09T08:21:43+00:00 by eval-builder 0.1.2.
3
+ Generated 2026-10-09T08:26:01+00:00 by eval-builder 0.1.2.
4
4
 
5
5
  ## Sources
6
6
 
@@ -106,18 +106,18 @@ Thresholds: flip rate <= 20% (cases with >= 3 trials, >= 10 cases), position con
106
106
  | judge | mode | verdict | flip rate | self-agreement | majority-of-3 stable | accuracy vs humans | kappa | position consistency | first-shown picked | padding helped |
107
107
  |---|---|---|---|---|---|---|---|---|---|---|
108
108
  | llama3.2-3b | pointwise | **unstable** | 47% [33%, 61%] (n=47) | 83% | 88% | 54% [35%, 72%] (n=24) | 0.12 [-0.26, 0.50] | n/a | n/a | 4% [1%, 14%] (n=47) |
109
- | qwen2.5-7b | pointwise | **trustworthy** | 9% [3%, 20%] (n=47) | 97% | 97% | 71% [51%, 85%] (n=24) | 0.44 [0.09, 0.79] | n/a | n/a | 0% [0%, 8%] (n=47) |
110
- | qwen2.5-7b-temp0 | pointwise | **trustworthy** | 0% [0%, 8%] (n=47) | 100% | 100% | 75% [55%, 88%] (n=24) | 0.50 [0.15, 0.85] | n/a | n/a | 0% [0%, 8%] (n=47) |
109
+ | qwen2.5-7b | pointwise | **misaligned** | 9% [3%, 20%] (n=47) | 97% | 97% | 71% [51%, 85%] (n=24) | 0.44 [0.09, 0.79] | n/a | n/a | 0% [0%, 8%] (n=47) |
110
+ | qwen2.5-7b-temp0 | pointwise | **misaligned** | 0% [0%, 8%] (n=47) | 100% | 100% | 75% [55%, 88%] (n=24) | 0.50 [0.15, 0.85] | n/a | n/a | 0% [0%, 8%] (n=47) |
111
111
 
112
112
  - **llama3.2-3b** (unstable): verdict changed across repeated trials on 47% of cases (limit 20%). agreement with human labels is low: kappa 0.12 (need 0.4), accuracy 54%.
113
- - **qwen2.5-7b** (trustworthy): stable (flip rate 9%), kappa 0.44 with humans.
114
- - **qwen2.5-7b-temp0** (trustworthy): stable (flip rate 0%), kappa 0.50 with humans.
113
+ - **qwen2.5-7b** (misaligned): accuracy 71% is no better than always answering 'fail' (75% on these labels), so kappa 0.44 alone doesn't show the judge adds anything.
114
+ - **qwen2.5-7b-temp0** (misaligned): accuracy 75% is no better than always answering 'fail' (75% on these labels), so kappa 0.50 alone doesn't show the judge adds anything.
115
115
 
116
116
  Human labels on judged cases: fail 18 of 24 (75%), pass 6 (25%). A judge that always gave the most common label would score 75% accuracy, which is the bar accuracy has to clear; kappa already corrects for it.
117
117
 
118
- - Warning: judge llama3.2-3b's accuracy (54%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said pass 13 of 24 (54%), fail 11 (46%)); kappa 0.12 [-0.26, 0.50] is the number that corrects for this. Look at the cases it got wrong before relying on it
119
- - Warning: judge qwen2.5-7b's accuracy (71%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said pass 13 of 24 (54%), fail 11 (46%)); kappa 0.44 [0.09, 0.79] is the number that corrects for this. Look at the cases it got wrong before relying on it
120
- - Warning: judge qwen2.5-7b-temp0's accuracy (75%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said fail 12 of 24 (50%), pass 12 (50%)); kappa 0.50 [0.15, 0.85] is the number that corrects for this. Look at the cases it got wrong before relying on it
118
+ - Warning: judge llama3.2-3b's accuracy (54%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said pass 13 of 24 (54%), fail 11 (46%); kappa 0.12 [-0.26, 0.50]). A judge has to beat that baseline to be called trustworthy; look at the cases it got wrong
119
+ - Warning: judge qwen2.5-7b's accuracy (71%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said pass 13 of 24 (54%), fail 11 (46%); kappa 0.44 [0.09, 0.79]). A judge has to beat that baseline to be called trustworthy; look at the cases it got wrong
120
+ - Warning: judge qwen2.5-7b-temp0's accuracy (75%) is no better than always answering 'fail' (75%) on these 24 labeled cases (the judge said fail 12 of 24 (50%), pass 12 (50%); kappa 0.50 [0.15, 0.85]). A judge has to beat that baseline to be called trustworthy; look at the cases it got wrong
121
121
 
122
122
  Judge calls made through the opt-in judge plugin (`judge-run`):
123
123
 
@@ -131,10 +131,9 @@ Judge calls made through the opt-in judge plugin (`judge-run`):
131
131
 
132
132
  | file | sha256 |
133
133
  |---|---|
134
- | exports/promptfoo/promptfooconfig.yaml | `23c43efb656ee17c` |
134
+ | exports/promptfoo/promptfooconfig.yaml | `07a874ff9a934316` |
135
135
  | exports/deepeval/dataset.json | `f6cf982eeef58e13` |
136
136
  | exports/deepeval/test_eval_builder.py | `05f445309c2983af` |
137
- | exports/deepeval/judge.json | `dd632220790c10ab` |
138
137
  | exports/inspect/dataset.jsonl | `63fb2547e1e81007` |
139
138
  | exports/inspect/task.py | `df284b6b6dedeb83` |
140
139
  | exports/jsonl/cases.jsonl | `d10eff47dcc67a11` |
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "eval-builder"
3
- version = "0.1.2"
3
+ version = "0.1.3"
4
4
  description = "Turn real LLM app logs into an eval suite and measure which LLM judges you can trust."
5
5
  readme = "README.md"
6
6
  license = "MIT"
@@ -1,3 +1,3 @@
1
1
  """eval-builder: turn real LLM app logs into an eval suite and check which judges you can trust."""
2
2
 
3
- __version__ = "0.1.2"
3
+ __version__ = "0.1.3"