judgekeeper 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (239) hide show
  1. judgekeeper-0.1.0/.github/workflows/ci.yml +110 -0
  2. judgekeeper-0.1.0/.github/workflows/pages.yml +63 -0
  3. judgekeeper-0.1.0/.github/workflows/release.yml +86 -0
  4. judgekeeper-0.1.0/.gitignore +18 -0
  5. judgekeeper-0.1.0/CHANGELOG.md +29 -0
  6. judgekeeper-0.1.0/CLAUDE.md +28 -0
  7. judgekeeper-0.1.0/LICENSE +21 -0
  8. judgekeeper-0.1.0/PKG-INFO +207 -0
  9. judgekeeper-0.1.0/README.md +175 -0
  10. judgekeeper-0.1.0/action.yml +173 -0
  11. judgekeeper-0.1.0/docs/RELEASING.md +140 -0
  12. judgekeeper-0.1.0/docs/examples/llmbar-haiku/report.html +989 -0
  13. judgekeeper-0.1.0/docs/examples/llmbar-haiku/report.json +13169 -0
  14. judgekeeper-0.1.0/docs/examples/workflows/judge-gate.yml +68 -0
  15. judgekeeper-0.1.0/docs/integrations/deepeval.md +38 -0
  16. judgekeeper-0.1.0/docs/integrations/inspect.md +40 -0
  17. judgekeeper-0.1.0/docs/integrations/langfuse.md +50 -0
  18. judgekeeper-0.1.0/docs/integrations/mlflow.md +55 -0
  19. judgekeeper-0.1.0/docs/integrations/promptfoo.md +50 -0
  20. judgekeeper-0.1.0/docs/reference.md +478 -0
  21. judgekeeper-0.1.0/docs/research/00-judge-reliability-synthesis.md +197 -0
  22. judgekeeper-0.1.0/docs/research/01-existing-solutions.md +158 -0
  23. judgekeeper-0.1.0/docs/research/02-research-and-open-source.md +159 -0
  24. judgekeeper-0.1.0/docs/research/03-practitioner-pain-and-demand.md +160 -0
  25. judgekeeper-0.1.0/docs/research/integrations/00-integration-and-adoption-plan.md +149 -0
  26. judgekeeper-0.1.0/docs/research/integrations/01-oss-library-formats.md +230 -0
  27. judgekeeper-0.1.0/docs/research/integrations/02-platform-exports-and-standards.md +244 -0
  28. judgekeeper-0.1.0/docs/research/integrations/03-no-framework-users-and-adoption.md +284 -0
  29. judgekeeper-0.1.0/docs/specs/session-01-core-validation.md +61 -0
  30. judgekeeper-0.1.0/docs/specs/session-02-ci-gate-action.md +64 -0
  31. judgekeeper-0.1.0/docs/specs/session-03-keys-providers-migration.md +109 -0
  32. judgekeeper-0.1.0/docs/specs/session-04-use-without-a-framework.md +88 -0
  33. judgekeeper-0.1.0/docs/specs/session-05-framework-file-readers.md +103 -0
  34. judgekeeper-0.1.0/docs/specs/session-06-mlflow-langfuse-readers.md +70 -0
  35. judgekeeper-0.1.0/docs/specs/session-07-launch-readiness.md +77 -0
  36. judgekeeper-0.1.0/docs/specs/session-08-website.md +77 -0
  37. judgekeeper-0.1.0/packages/pytest-judgekeeper/README.md +4 -0
  38. judgekeeper-0.1.0/packages/pytest-judgekeeper/pyproject.toml +31 -0
  39. judgekeeper-0.1.0/prompts/pairwise.md +21 -0
  40. judgekeeper-0.1.0/prompts/single.md +12 -0
  41. judgekeeper-0.1.0/pyproject.toml +49 -0
  42. judgekeeper-0.1.0/scripts/build_pages.py +92 -0
  43. judgekeeper-0.1.0/scripts/llmbar_haiku_demo.sh +25 -0
  44. judgekeeper-0.1.0/scripts/make_demo_data.py +102 -0
  45. judgekeeper-0.1.0/scripts/migration_demo.sh +45 -0
  46. judgekeeper-0.1.0/scripts/render_reference.py +134 -0
  47. judgekeeper-0.1.0/skills/judgekeeper/SKILL.md +125 -0
  48. judgekeeper-0.1.0/src/judgekeeper/__init__.py +22 -0
  49. judgekeeper-0.1.0/src/judgekeeper/anchors.py +146 -0
  50. judgekeeper-0.1.0/src/judgekeeper/attribute.py +216 -0
  51. judgekeeper-0.1.0/src/judgekeeper/cli.py +740 -0
  52. judgekeeper-0.1.0/src/judgekeeper/config.py +41 -0
  53. judgekeeper-0.1.0/src/judgekeeper/custom.py +236 -0
  54. judgekeeper-0.1.0/src/judgekeeper/datasets/__init__.py +1 -0
  55. judgekeeper-0.1.0/src/judgekeeper/datasets/llmbar.py +79 -0
  56. judgekeeper-0.1.0/src/judgekeeper/demo.py +32 -0
  57. judgekeeper-0.1.0/src/judgekeeper/demo_data/NOTICE +3 -0
  58. judgekeeper-0.1.0/src/judgekeeper/demo_data/anchors.jsonl +100 -0
  59. judgekeeper-0.1.0/src/judgekeeper/demo_data/anchors.manifest.json +14 -0
  60. judgekeeper-0.1.0/src/judgekeeper/demo_data/runs/run-01.jsonl +101 -0
  61. judgekeeper-0.1.0/src/judgekeeper/demo_data/runs/run-02.jsonl +101 -0
  62. judgekeeper-0.1.0/src/judgekeeper/demo_data/runs/run-03.jsonl +101 -0
  63. judgekeeper-0.1.0/src/judgekeeper/fingerprint.py +91 -0
  64. judgekeeper-0.1.0/src/judgekeeper/gate.py +378 -0
  65. judgekeeper-0.1.0/src/judgekeeper/html_report.py +279 -0
  66. judgekeeper-0.1.0/src/judgekeeper/judging.py +79 -0
  67. judgekeeper-0.1.0/src/judgekeeper/judgments.py +138 -0
  68. judgekeeper-0.1.0/src/judgekeeper/label.py +579 -0
  69. judgekeeper-0.1.0/src/judgekeeper/metrics.py +191 -0
  70. judgekeeper-0.1.0/src/judgekeeper/migrate.py +455 -0
  71. judgekeeper-0.1.0/src/judgekeeper/normalise.py +229 -0
  72. judgekeeper-0.1.0/src/judgekeeper/prompts.py +66 -0
  73. judgekeeper-0.1.0/src/judgekeeper/pytest_plugin.py +143 -0
  74. judgekeeper-0.1.0/src/judgekeeper/readers/__init__.py +124 -0
  75. judgekeeper-0.1.0/src/judgekeeper/readers/deepeval.py +100 -0
  76. judgekeeper-0.1.0/src/judgekeeper/readers/inspect_logs.py +153 -0
  77. judgekeeper-0.1.0/src/judgekeeper/readers/langfuse_api.py +329 -0
  78. judgekeeper-0.1.0/src/judgekeeper/readers/mlflow_store.py +251 -0
  79. judgekeeper-0.1.0/src/judgekeeper/readers/promptfoo.py +191 -0
  80. judgekeeper-0.1.0/src/judgekeeper/records.py +566 -0
  81. judgekeeper-0.1.0/src/judgekeeper/redact.py +66 -0
  82. judgekeeper-0.1.0/src/judgekeeper/report.py +417 -0
  83. judgekeeper-0.1.0/src/judgekeeper/runners/__init__.py +6 -0
  84. judgekeeper-0.1.0/src/judgekeeper/runners/anthropic.py +53 -0
  85. judgekeeper-0.1.0/src/judgekeeper/runners/base.py +125 -0
  86. judgekeeper-0.1.0/src/judgekeeper/runners/openai.py +51 -0
  87. judgekeeper-0.1.0/src/judgekeeper/runners/replay.py +28 -0
  88. judgekeeper-0.1.0/src/judgekeeper/table.py +363 -0
  89. judgekeeper-0.1.0/tests/__init__.py +0 -0
  90. judgekeeper-0.1.0/tests/conftest.py +44 -0
  91. judgekeeper-0.1.0/tests/fixtures/check/results.csv +31 -0
  92. judgekeeper-0.1.0/tests/fixtures/check/scores.jsonl +10 -0
  93. judgekeeper-0.1.0/tests/fixtures/deepeval/README.md +22 -0
  94. judgekeeper-0.1.0/tests/fixtures/deepeval/labels.csv +10 -0
  95. judgekeeper-0.1.0/tests/fixtures/deepeval/real/README.md +19 -0
  96. judgekeeper-0.1.0/tests/fixtures/deepeval/real/labels.csv +7 -0
  97. judgekeeper-0.1.0/tests/fixtures/deepeval/real/make_runs.py +71 -0
  98. judgekeeper-0.1.0/tests/fixtures/deepeval/real/results/test_run_20261001_143217.json +1 -0
  99. judgekeeper-0.1.0/tests/fixtures/deepeval/real/results/test_run_20261001_143219.json +1 -0
  100. judgekeeper-0.1.0/tests/fixtures/deepeval/real/results/test_run_20261001_143220.json +1 -0
  101. judgekeeper-0.1.0/tests/fixtures/deepeval/results/test_run_20260930_100000.json +275 -0
  102. judgekeeper-0.1.0/tests/fixtures/deepeval/results/test_run_20260930_110000.json +275 -0
  103. judgekeeper-0.1.0/tests/fixtures/deepeval/results/test_run_20260930_120000.json +275 -0
  104. judgekeeper-0.1.0/tests/fixtures/exec/bare.py +6 -0
  105. judgekeeper-0.1.0/tests/fixtures/exec/fails_some.py +9 -0
  106. judgekeeper-0.1.0/tests/fixtures/exec/json_verdict.py +7 -0
  107. judgekeeper-0.1.0/tests/fixtures/gate/anchors-changed.json +70 -0
  108. judgekeeper-0.1.0/tests/fixtures/gate/baseline.json +70 -0
  109. judgekeeper-0.1.0/tests/fixtures/gate/best-run-passes.json +70 -0
  110. judgekeeper-0.1.0/tests/fixtures/gate/kappa-drop-inside-band.json +70 -0
  111. judgekeeper-0.1.0/tests/fixtures/gate/kappa-drop-outside-band.json +70 -0
  112. judgekeeper-0.1.0/tests/fixtures/gate/pass.json +70 -0
  113. judgekeeper-0.1.0/tests/fixtures/gate/passes-everything.json +54 -0
  114. judgekeeper-0.1.0/tests/fixtures/gate/snapshot-changed.json +70 -0
  115. judgekeeper-0.1.0/tests/fixtures/gate/two-runs.json +60 -0
  116. judgekeeper-0.1.0/tests/fixtures/inspect/README.md +21 -0
  117. judgekeeper-0.1.0/tests/fixtures/inspect/labels.csv +6 -0
  118. judgekeeper-0.1.0/tests/fixtures/inspect/logs/2026-09-30T10-00-00+00-00_qa_fixture.json +946 -0
  119. judgekeeper-0.1.0/tests/fixtures/inspect/real/README.md +21 -0
  120. judgekeeper-0.1.0/tests/fixtures/inspect/real/labels.csv +7 -0
  121. judgekeeper-0.1.0/tests/fixtures/inspect/real/logs/model_graded_qa.json +9509 -0
  122. judgekeeper-0.1.0/tests/fixtures/inspect/real/make_log.py +58 -0
  123. judgekeeper-0.1.0/tests/fixtures/langfuse/README.md +17 -0
  124. judgekeeper-0.1.0/tests/fixtures/langfuse/evaluators-default-model.json +52 -0
  125. judgekeeper-0.1.0/tests/fixtures/langfuse/evaluators.json +105 -0
  126. judgekeeper-0.1.0/tests/fixtures/langfuse/make_responses.py +121 -0
  127. judgekeeper-0.1.0/tests/fixtures/langfuse/scores-page-1.json +113 -0
  128. judgekeeper-0.1.0/tests/fixtures/langfuse/scores-page-2.json +113 -0
  129. judgekeeper-0.1.0/tests/fixtures/langfuse/scores-page-3.json +70 -0
  130. judgekeeper-0.1.0/tests/fixtures/make_framework_fixtures.py +284 -0
  131. judgekeeper-0.1.0/tests/fixtures/migrate/anchors.jsonl +10 -0
  132. judgekeeper-0.1.0/tests/fixtures/migrate/anchors.manifest.json +13 -0
  133. judgekeeper-0.1.0/tests/fixtures/migrate/make_fixtures.py +95 -0
  134. judgekeeper-0.1.0/tests/fixtures/migrate/new-better/run-01.jsonl +11 -0
  135. judgekeeper-0.1.0/tests/fixtures/migrate/new-better/run-02.jsonl +11 -0
  136. judgekeeper-0.1.0/tests/fixtures/migrate/new-better/run-03.jsonl +11 -0
  137. judgekeeper-0.1.0/tests/fixtures/migrate/new-different/run-01.jsonl +11 -0
  138. judgekeeper-0.1.0/tests/fixtures/migrate/new-different/run-02.jsonl +11 -0
  139. judgekeeper-0.1.0/tests/fixtures/migrate/new-different/run-03.jsonl +11 -0
  140. judgekeeper-0.1.0/tests/fixtures/migrate/new-equivalent/run-01.jsonl +11 -0
  141. judgekeeper-0.1.0/tests/fixtures/migrate/new-equivalent/run-02.jsonl +11 -0
  142. judgekeeper-0.1.0/tests/fixtures/migrate/new-equivalent/run-03.jsonl +11 -0
  143. judgekeeper-0.1.0/tests/fixtures/migrate/new-worse/run-01.jsonl +11 -0
  144. judgekeeper-0.1.0/tests/fixtures/migrate/new-worse/run-02.jsonl +11 -0
  145. judgekeeper-0.1.0/tests/fixtures/migrate/new-worse/run-03.jsonl +11 -0
  146. judgekeeper-0.1.0/tests/fixtures/migrate/old/run-01.jsonl +11 -0
  147. judgekeeper-0.1.0/tests/fixtures/migrate/old/run-02.jsonl +11 -0
  148. judgekeeper-0.1.0/tests/fixtures/migrate/old/run-03.jsonl +11 -0
  149. judgekeeper-0.1.0/tests/fixtures/migrate/old-drifted/run-01.jsonl +11 -0
  150. judgekeeper-0.1.0/tests/fixtures/migrate/old-drifted/run-02.jsonl +11 -0
  151. judgekeeper-0.1.0/tests/fixtures/migrate/old-drifted/run-03.jsonl +11 -0
  152. judgekeeper-0.1.0/tests/fixtures/migrate/old-new-snapshot/run-01.jsonl +11 -0
  153. judgekeeper-0.1.0/tests/fixtures/migrate/old-new-snapshot/run-02.jsonl +11 -0
  154. judgekeeper-0.1.0/tests/fixtures/migrate/old-new-snapshot/run-03.jsonl +11 -0
  155. judgekeeper-0.1.0/tests/fixtures/migrate/old-noisy/run-01.jsonl +11 -0
  156. judgekeeper-0.1.0/tests/fixtures/migrate/old-noisy/run-02.jsonl +11 -0
  157. judgekeeper-0.1.0/tests/fixtures/migrate/old-noisy/run-03.jsonl +11 -0
  158. judgekeeper-0.1.0/tests/fixtures/migrate/old-perfect/run-01.jsonl +11 -0
  159. judgekeeper-0.1.0/tests/fixtures/migrate/old-perfect/run-02.jsonl +11 -0
  160. judgekeeper-0.1.0/tests/fixtures/migrate/old-perfect/run-03.jsonl +11 -0
  161. judgekeeper-0.1.0/tests/fixtures/migrate/old-rerun/run-01.jsonl +11 -0
  162. judgekeeper-0.1.0/tests/fixtures/migrate/old-rerun/run-02.jsonl +11 -0
  163. judgekeeper-0.1.0/tests/fixtures/migrate/old-rerun/run-03.jsonl +11 -0
  164. judgekeeper-0.1.0/tests/fixtures/mlflow/README.md +30 -0
  165. judgekeeper-0.1.0/tests/fixtures/mlflow/make_store.py +121 -0
  166. judgekeeper-0.1.0/tests/fixtures/pairwise/anchors.jsonl +8 -0
  167. judgekeeper-0.1.0/tests/fixtures/pairwise/anchors.manifest.json +13 -0
  168. judgekeeper-0.1.0/tests/fixtures/pairwise/baseline.json +289 -0
  169. judgekeeper-0.1.0/tests/fixtures/pairwise/replay-agree.jsonl +9 -0
  170. judgekeeper-0.1.0/tests/fixtures/pairwise/runs/run-01.jsonl +9 -0
  171. judgekeeper-0.1.0/tests/fixtures/pairwise/runs/run-02.jsonl +9 -0
  172. judgekeeper-0.1.0/tests/fixtures/pairwise/runs/run-03.jsonl +9 -0
  173. judgekeeper-0.1.0/tests/fixtures/promptfoo/README.md +21 -0
  174. judgekeeper-0.1.0/tests/fixtures/promptfoo/real/README.md +18 -0
  175. judgekeeper-0.1.0/tests/fixtures/promptfoo/real/answerer.py +10 -0
  176. judgekeeper-0.1.0/tests/fixtures/promptfoo/real/grader.py +23 -0
  177. judgekeeper-0.1.0/tests/fixtures/promptfoo/real/labels.csv +7 -0
  178. judgekeeper-0.1.0/tests/fixtures/promptfoo/real/make.sh +11 -0
  179. judgekeeper-0.1.0/tests/fixtures/promptfoo/real/promptfooconfig.yaml +20 -0
  180. judgekeeper-0.1.0/tests/fixtures/promptfoo/real/results.json +2574 -0
  181. judgekeeper-0.1.0/tests/fixtures/promptfoo/results.json +1781 -0
  182. judgekeeper-0.1.0/tests/fixtures/records/README.md +6 -0
  183. judgekeeper-0.1.0/tests/fixtures/records/export.csv +11 -0
  184. judgekeeper-0.1.0/tests/judges_for_tests.py +10 -0
  185. judgekeeper-0.1.0/tests/langfuse_stub.py +95 -0
  186. judgekeeper-0.1.0/tests/test_anchors.py +115 -0
  187. judgekeeper-0.1.0/tests/test_attribute.py +215 -0
  188. judgekeeper-0.1.0/tests/test_check.py +211 -0
  189. judgekeeper-0.1.0/tests/test_cli.py +126 -0
  190. judgekeeper-0.1.0/tests/test_cli_gate.py +193 -0
  191. judgekeeper-0.1.0/tests/test_custom_judge.py +261 -0
  192. judgekeeper-0.1.0/tests/test_demo.py +46 -0
  193. judgekeeper-0.1.0/tests/test_endpoints.py +251 -0
  194. judgekeeper-0.1.0/tests/test_fingerprint_unknown.py +114 -0
  195. judgekeeper-0.1.0/tests/test_gate.py +367 -0
  196. judgekeeper-0.1.0/tests/test_import_deepeval.py +159 -0
  197. judgekeeper-0.1.0/tests/test_import_inspect.py +161 -0
  198. judgekeeper-0.1.0/tests/test_import_langfuse.py +312 -0
  199. judgekeeper-0.1.0/tests/test_import_mlflow.py +257 -0
  200. judgekeeper-0.1.0/tests/test_import_promptfoo.py +187 -0
  201. judgekeeper-0.1.0/tests/test_import_real.py +174 -0
  202. judgekeeper-0.1.0/tests/test_label_server.py +333 -0
  203. judgekeeper-0.1.0/tests/test_labels.py +113 -0
  204. judgekeeper-0.1.0/tests/test_leak.py +292 -0
  205. judgekeeper-0.1.0/tests/test_llmbar.py +54 -0
  206. judgekeeper-0.1.0/tests/test_metrics.py +128 -0
  207. judgekeeper-0.1.0/tests/test_migrate.py +371 -0
  208. judgekeeper-0.1.0/tests/test_mlflow_warnings.py +42 -0
  209. judgekeeper-0.1.0/tests/test_normalise.py +158 -0
  210. judgekeeper-0.1.0/tests/test_prompts.py +63 -0
  211. judgekeeper-0.1.0/tests/test_pytest_plugin.py +202 -0
  212. judgekeeper-0.1.0/tests/test_records.py +258 -0
  213. judgekeeper-0.1.0/tests/test_redact.py +141 -0
  214. judgekeeper-0.1.0/tests/test_release.py +298 -0
  215. judgekeeper-0.1.0/tests/test_repo.py +37 -0
  216. judgekeeper-0.1.0/tests/test_report_headline.py +96 -0
  217. judgekeeper-0.1.0/tests/test_runners.py +154 -0
  218. judgekeeper-0.1.0/tests/test_skill.py +103 -0
  219. judgekeeper-0.1.0/tests/test_smoke.py +14 -0
  220. judgekeeper-0.1.0/tests/test_validate.py +150 -0
  221. judgekeeper-0.1.0/tests/test_website.py +366 -0
  222. judgekeeper-0.1.0/tests/test_website_tutorial.py +151 -0
  223. judgekeeper-0.1.0/tests/website_pages.py +141 -0
  224. judgekeeper-0.1.0/website/README.md +60 -0
  225. judgekeeper-0.1.0/website/assets/logo.svg +9 -0
  226. judgekeeper-0.1.0/website/assets/metrics.js +67 -0
  227. judgekeeper-0.1.0/website/assets/site.js +259 -0
  228. judgekeeper-0.1.0/website/assets/style.css +530 -0
  229. judgekeeper-0.1.0/website/index.html +620 -0
  230. judgekeeper-0.1.0/website/reference.html +380 -0
  231. judgekeeper-0.1.0/website/setup.html +238 -0
  232. judgekeeper-0.1.0/website/tutorial/careful_judge.py +25 -0
  233. judgekeeper-0.1.0/website/tutorial/items.csv +11 -0
  234. judgekeeper-0.1.0/website/tutorial/labels-filled.csv +11 -0
  235. judgekeeper-0.1.0/website/tutorial/lazy_judge.py +16 -0
  236. judgekeeper-0.1.0/website/tutorial/promptfoo/labels.csv +7 -0
  237. judgekeeper-0.1.0/website/tutorial/promptfoo/results.json +2574 -0
  238. judgekeeper-0.1.0/website/tutorial/results.csv +31 -0
  239. judgekeeper-0.1.0/website/tutorial.html +192 -0
@@ -0,0 +1,110 @@
1
+ name: CI
2
+
3
+ on:
4
+ push:
5
+ pull_request:
6
+
7
+ permissions:
8
+ contents: read
9
+
10
+ jobs:
11
+ test:
12
+ runs-on: ubuntu-latest
13
+ strategy:
14
+ fail-fast: false
15
+ matrix:
16
+ python-version: ["3.11", "3.12"]
17
+ steps:
18
+ - uses: actions/checkout@v4
19
+ - uses: actions/setup-python@v5
20
+ with:
21
+ python-version: ${{ matrix.python-version }}
22
+ - name: Install
23
+ # The mlflow extra lets tests/test_import_mlflow.py build and read a real MLflow store.
24
+ run: |
25
+ python -m pip install --upgrade pip
26
+ pip install -e ".[dev,mlflow]"
27
+ - name: Test
28
+ run: pytest
29
+ - name: Lint
30
+ run: ruff check src tests
31
+
32
+ # Builds both distributions, installs the wheels into a fresh venv and runs them from there,
33
+ # so a packaging mistake (missing demo_data or NOTICE, a broken entry point) fails here
34
+ # rather than on release day. The same steps run in release.yml before publishing.
35
+ package:
36
+ runs-on: ubuntu-latest
37
+ steps:
38
+ - uses: actions/checkout@v4
39
+ - uses: actions/setup-python@v5
40
+ with:
41
+ python-version: "3.12"
42
+ - name: Build both distributions
43
+ run: |
44
+ python -m pip install --upgrade pip build twine
45
+ python -m build --outdir dist .
46
+ python -m build --outdir dist packages/pytest-judgekeeper
47
+ twine check --strict dist/*
48
+ - name: Install the wheel into a fresh venv
49
+ run: |
50
+ python -m venv "$RUNNER_TEMP/venv"
51
+ "$RUNNER_TEMP/venv/bin/pip" install dist/judgekeeper-*.whl
52
+ - name: Run the installed judgekeeper outside the checkout
53
+ working-directory: ${{ runner.temp }}
54
+ run: |
55
+ set -x
56
+ venv/bin/judgekeeper --help
57
+ venv/bin/judgekeeper demo --out demo
58
+ test -f demo/report.html -a -f demo/report.json
59
+ venv/bin/python -c 'import judgekeeper.demo as d; assert (d.DEMO_DATA / "NOTICE").is_file()'
60
+ - name: Run the test suite against the installed wheels
61
+ run: |
62
+ "$RUNNER_TEMP/venv/bin/pip" install --find-links dist dist/pytest_judgekeeper-*.whl \
63
+ "pytest>=8" "pyyaml>=6"
64
+ "$RUNNER_TEMP/venv/bin/python" -c 'import judgekeeper, sys; assert "site-packages" in judgekeeper.__file__, judgekeeper.__file__'
65
+ "$RUNNER_TEMP/venv/bin/python" -m pytest -p no:cacheprovider
66
+
67
+ # Runs action.yml from this checkout with the replay runner: no API key needed.
68
+ action:
69
+ runs-on: ubuntu-latest
70
+ steps:
71
+ - uses: actions/checkout@v4
72
+
73
+ - name: Gate a judge that agrees with every human label (expect PASS)
74
+ id: agree
75
+ uses: ./
76
+ with:
77
+ anchors: tests/fixtures/pairwise/anchors.jsonl
78
+ runner: replay
79
+ fixture: tests/fixtures/pairwise/replay-agree.jsonl
80
+ baseline: tests/fixtures/pairwise/baseline.json
81
+ out-dir: gate-agree
82
+ artifact-name: judgekeeper-gate-agree
83
+
84
+ - name: Gate a judge with kappa 0.5 (expect FAIL and exit 1)
85
+ id: disagree
86
+ continue-on-error: true
87
+ uses: ./
88
+ with:
89
+ anchors: tests/fixtures/pairwise/anchors.jsonl
90
+ runner: replay
91
+ fixture: tests/fixtures/pairwise/runs/run-01.jsonl
92
+ baseline: tests/fixtures/pairwise/baseline.json
93
+ out-dir: gate-disagree
94
+ artifact-name: judgekeeper-gate-disagree
95
+
96
+ - name: Check the action's status and exit code
97
+ env:
98
+ AGREE_STATUS: ${{ steps.agree.outputs.status }}
99
+ AGREE_CODE: ${{ steps.agree.outputs.exit-code }}
100
+ DISAGREE_OUTCOME: ${{ steps.disagree.outcome }}
101
+ DISAGREE_STATUS: ${{ steps.disagree.outputs.status }}
102
+ DISAGREE_CODE: ${{ steps.disagree.outputs.exit-code }}
103
+ run: |
104
+ set -x
105
+ test "$AGREE_STATUS" = PASS
106
+ test "$AGREE_CODE" = 0
107
+ test "$DISAGREE_OUTCOME" = failure
108
+ test "$DISAGREE_STATUS" = FAIL
109
+ test "$DISAGREE_CODE" = 1
110
+ test -f gate-agree/report.html -a -f gate-agree/report.json -a -f gate-agree/gate.json
@@ -0,0 +1,63 @@
1
+ # Publishes the website (website/) to GitHub Pages at the site root, with the HTML reports
2
+ # under docs/examples/ at examples/<name>/, e.g.
3
+ # https://<owner>.github.io/judgekeeper/ and https://<owner>.github.io/judgekeeper/examples/llmbar-haiku/
4
+ # Inert until the maintainer enables Pages (Settings > Pages > Source: GitHub Actions; see
5
+ # docs/RELEASING.md): until then the build job sees no Pages site and the deploy job is skipped.
6
+ name: Pages
7
+
8
+ on:
9
+ push:
10
+ branches: [main]
11
+ workflow_dispatch:
12
+
13
+ permissions:
14
+ contents: read
15
+
16
+ concurrency:
17
+ group: pages
18
+ cancel-in-progress: false
19
+
20
+ jobs:
21
+ build:
22
+ runs-on: ubuntu-latest
23
+ permissions:
24
+ contents: read
25
+ pages: read
26
+ outputs:
27
+ enabled: ${{ steps.check.outputs.enabled }}
28
+ steps:
29
+ - name: Is Pages enabled for this repository?
30
+ id: check
31
+ env:
32
+ GH_TOKEN: ${{ github.token }}
33
+ run: |
34
+ if gh api "repos/$GITHUB_REPOSITORY/pages" --silent 2>/dev/null; then
35
+ echo "enabled=true" >> "$GITHUB_OUTPUT"
36
+ else
37
+ echo "enabled=false" >> "$GITHUB_OUTPUT"
38
+ echo "Pages is not enabled for this repository: nothing to publish." >> "$GITHUB_STEP_SUMMARY"
39
+ fi
40
+ - uses: actions/checkout@v4
41
+ - uses: actions/setup-python@v5
42
+ with:
43
+ python-version: "3.12"
44
+ - name: Build the site
45
+ run: python scripts/build_pages.py website docs/examples _site
46
+ - if: steps.check.outputs.enabled == 'true'
47
+ uses: actions/upload-pages-artifact@v3
48
+ with:
49
+ path: _site
50
+
51
+ deploy:
52
+ needs: build
53
+ if: needs.build.outputs.enabled == 'true'
54
+ runs-on: ubuntu-latest
55
+ permissions:
56
+ pages: write
57
+ id-token: write
58
+ environment:
59
+ name: github-pages
60
+ url: ${{ steps.deployment.outputs.page_url }}
61
+ steps:
62
+ - id: deployment
63
+ uses: actions/deploy-pages@v4
@@ -0,0 +1,86 @@
1
+ # Publishes judgekeeper and pytest-judgekeeper to PyPI when a GitHub release is published.
2
+ # Nothing else triggers it. PyPI Trusted Publishing (OIDC) authenticates the upload: there is
3
+ # no API token anywhere. One-time setup by the maintainer: docs/RELEASING.md.
4
+ name: Release
5
+
6
+ on:
7
+ release:
8
+ types: [published]
9
+
10
+ permissions:
11
+ contents: read
12
+
13
+ jobs:
14
+ build:
15
+ runs-on: ubuntu-latest
16
+ steps:
17
+ - uses: actions/checkout@v4
18
+ - uses: actions/setup-python@v5
19
+ with:
20
+ python-version: "3.12"
21
+ - name: Install build tools
22
+ run: python -m pip install --upgrade pip build twine
23
+ - name: Build both distributions
24
+ run: |
25
+ python -m build --outdir dist .
26
+ python -m build --outdir dist packages/pytest-judgekeeper
27
+ - name: Check the metadata PyPI will render
28
+ run: twine check --strict dist/*
29
+ - name: Check the tag matches the version
30
+ env:
31
+ TAG: ${{ github.event.release.tag_name }}
32
+ run: |
33
+ version=$(python -c 'import tomllib; print(tomllib.load(open("pyproject.toml", "rb"))["project"]["version"])')
34
+ test "$TAG" = "v$version"
35
+ ls dist/judgekeeper-"$version"-py3-none-any.whl dist/pytest_judgekeeper-"$version"-py3-none-any.whl
36
+ - name: Run the test suite against the built wheel in a clean venv
37
+ run: |
38
+ python -m venv "$RUNNER_TEMP/venv"
39
+ "$RUNNER_TEMP/venv/bin/pip" install --find-links dist dist/judgekeeper-*.whl \
40
+ dist/pytest_judgekeeper-*.whl "pytest>=8" "pyyaml>=6"
41
+ # src/ is not on the path: every test imports the installed wheel.
42
+ "$RUNNER_TEMP/venv/bin/python" -c 'import judgekeeper, sys; assert "site-packages" in judgekeeper.__file__, judgekeeper.__file__'
43
+ "$RUNNER_TEMP/venv/bin/python" -m pytest -p no:cacheprovider
44
+ - uses: actions/upload-artifact@v4
45
+ with:
46
+ name: dist
47
+ path: dist/
48
+ if-no-files-found: error
49
+
50
+ # One upload job per PyPI project, each behind its own environment (pypi, pypi-plugin):
51
+ # PyPI refuses two pending Trusted Publishers with the same repository, workflow and
52
+ # environment, and both projects are new.
53
+ publish:
54
+ needs: build
55
+ runs-on: ubuntu-latest
56
+ environment:
57
+ name: pypi
58
+ url: https://pypi.org/project/judgekeeper/
59
+ permissions:
60
+ id-token: write # Trusted Publishing: PyPI exchanges this OIDC token for a short upload token
61
+ steps:
62
+ - uses: actions/download-artifact@v4
63
+ with:
64
+ name: dist
65
+ path: dist/
66
+ - name: Keep only judgekeeper
67
+ run: rm dist/pytest_judgekeeper-*
68
+ - uses: pypa/gh-action-pypi-publish@release/v1
69
+
70
+ # pytest-judgekeeper pins judgekeeper==<version>, so it goes up second.
71
+ publish-plugin:
72
+ needs: [build, publish]
73
+ runs-on: ubuntu-latest
74
+ environment:
75
+ name: pypi-plugin
76
+ url: https://pypi.org/project/pytest-judgekeeper/
77
+ permissions:
78
+ id-token: write
79
+ steps:
80
+ - uses: actions/download-artifact@v4
81
+ with:
82
+ name: dist
83
+ path: dist/
84
+ - name: Keep only pytest-judgekeeper
85
+ run: rm dist/judgekeeper-*
86
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,18 @@
1
+ __pycache__/
2
+ *.pyc
3
+ .venv/
4
+ venv/
5
+ dist/
6
+ build/
7
+ *.egg-info/
8
+ .pytest_cache/
9
+ .ruff_cache/
10
+ .env
11
+ .DS_Store
12
+ # judge outputs and reports are reproducible artifacts; commit only curated examples under docs/
13
+ /runs/
14
+ /reports/
15
+ anchors/
16
+ # `judgekeeper demo` and `judgekeeper check` write here by default
17
+ /judgekeeper-demo/
18
+ /judgekeeper-report/
@@ -0,0 +1,29 @@
1
+ # Changelog
2
+
3
+ ## 0.1.0 (2026-10-02)
4
+
5
+ The first release: alpha, so commands, flags and `report.json` keys may still change.
6
+
7
+ - Validate an LLM judge against a frozen human-labeled anchor set: TPR and TNR with 95%
8
+ intervals, Cohen's kappa, the noise floor across repeated runs, AB/BA position bias, per-slice
9
+ numbers and label-quality warnings, in `report.json` and a self-contained `report.html`.
10
+ - Judges: built-in Anthropic and OpenAI-compatible runners, your own Python function
11
+ (`check_judge`, `--callable`), any program (`--exec`), and a replay runner for tests. Every
12
+ judgment carries a judge fingerprint; API keys are read from environment variables only and
13
+ scrubbed from everything judgekeeper prints or writes.
14
+ - No framework needed: `check` for a CSV or JSONL of verdicts and labels, `demo` with bundled
15
+ synthetic data, `template` and `import-labels` for spreadsheet labels, and `label`, a local
16
+ labeling page.
17
+ - Readers for promptfoo, DeepEval, Inspect AI, MLflow and Langfuse results, and the ScoreRecord
18
+ import and export format.
19
+ - Keep checking: `baseline`, `gate` (PASS, FAIL, FLAKY, JUDGE_CHANGED, ANCHORS_CHANGED, never
20
+ failing inside the noise band), a GitHub Action, a pytest plugin (also published as
21
+ `pytest-judgekeeper`), `migrate` to compare an old and a new judge, and `attribute` to tell
22
+ judge drift from a change in your system.
23
+ - A skill file for coding agents: `skills/judgekeeper/SKILL.md`.
24
+ - A static website in `website/`: what judgekeeper does, step by step, with an interactive
25
+ confusion matrix, a setup guide for API keys, a hands-on tutorial and the reference.
26
+ Published by GitHub Pages once enabled; preview with `python3 -m http.server --directory website`.
27
+
28
+ Measured so far: one live run, LLMBar judged by `claude-haiku-4-5-20251001`
29
+ (`docs/examples/llmbar-haiku/`). The live judge-migration demo has not been run yet.
@@ -0,0 +1,28 @@
1
+ # judgekeeper
2
+
3
+ Open-source Python toolkit that validates, monitors and migrates an LLM-as-judge against a frozen human-labeled anchor set. Read `README.md` for positioning and `docs/research/00-judge-reliability-synthesis.md` for the evidence behind every design choice.
4
+
5
+ ## How work is organised
6
+
7
+ - Each build session has a spec in `docs/specs/`. Do the spec in front of you, nothing beyond it. Non-goals in a spec are real.
8
+ - Package code lives in `src/judgekeeper/`. Tests live in `tests/`.
9
+ - Public human-labeled datasets used for demos are downloaded by loader code, never committed. Curated example reports go under `docs/examples/`.
10
+
11
+ ## Commands
12
+
13
+ ```
14
+ uv venv && source .venv/bin/activate
15
+ uv pip install -e ".[dev]"
16
+ pytest
17
+ ruff check src tests
18
+ ```
19
+
20
+ ## Rules
21
+
22
+ - Tests first. Unit tests never call an LLM API. Judge calls go behind a runner interface with a replay runner that reads recorded judgments from JSONL fixtures.
23
+ - Every judgment written to disk carries a judge fingerprint (provider, model id, snapshot, prompt hash, rubric version, temperature, timestamp). No exceptions.
24
+ - Report kappa, TPR and TNR. Never report raw agreement on its own.
25
+ - Keep the core dependency-free where practical. Metrics are implemented in plain Python; numpy is acceptable, scikit-learn is not required.
26
+ - Anthropic and OpenAI runners read API keys from `ANTHROPIC_API_KEY` and `OPENAI_API_KEY`. Never write keys to files.
27
+ - Do not invent numbers in README or docs. Every figure must come from a run whose report is committed under `docs/examples/`.
28
+ - Commit messages: short imperative subject, body explains why.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Sathvik Thota
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,207 @@
1
+ Metadata-Version: 2.5
2
+ Name: judgekeeper
3
+ Version: 0.1.0
4
+ Summary: Validate, monitor and migrate your LLM-as-judge against a frozen human reference set.
5
+ Project-URL: Homepage, https://github.com/judgekeeper/judgekeeper
6
+ Project-URL: Changelog, https://github.com/judgekeeper/judgekeeper/blob/main/CHANGELOG.md
7
+ Author: Sathvik Thota
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: calibration,drift,evaluation,llm,llm-as-judge
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Framework :: Pytest
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Topic :: Software Development :: Testing
16
+ Requires-Python: >=3.11
17
+ Provides-Extra: anthropic
18
+ Requires-Dist: anthropic>=0.40; extra == 'anthropic'
19
+ Provides-Extra: dev
20
+ Requires-Dist: build>=1; extra == 'dev'
21
+ Requires-Dist: hatchling>=1.26; extra == 'dev'
22
+ Requires-Dist: pytest>=8; extra == 'dev'
23
+ Requires-Dist: pyyaml>=6; extra == 'dev'
24
+ Requires-Dist: ruff>=0.6; extra == 'dev'
25
+ Provides-Extra: inspect
26
+ Requires-Dist: inspect-ai>=0.3; extra == 'inspect'
27
+ Provides-Extra: mlflow
28
+ Requires-Dist: mlflow>=3.1; extra == 'mlflow'
29
+ Provides-Extra: openai
30
+ Requires-Dist: openai>=1.0; extra == 'openai'
31
+ Description-Content-Type: text/markdown
32
+
33
+ # judgekeeper
34
+
35
+ **Checks whether your LLM-as-judge agrees with human labels, and keeps checking:** TPR, TNR and kappa against a frozen human-labeled set, run-to-run noise, position bias, drift and a CI gate.
36
+
37
+ ```
38
+ pip install judgekeeper
39
+ # before the first PyPI release: pip install "judgekeeper @ git+https://github.com/judgekeeper/judgekeeper"
40
+ judgekeeper demo
41
+ ```
42
+
43
+ `demo` validates a recorded judge on 100 bundled synthetic items: no API key, no network, a few seconds. It prints TPR, TNR and kappa and writes `judgekeeper-demo/report.html`. Without installing: `uvx judgekeeper demo` (before the PyPI release: `uvx --from git+https://github.com/judgekeeper/judgekeeper judgekeeper demo`).
44
+
45
+ See a real report without installing anything: [LLMBar judged by Claude Haiku 4.5](docs/examples/llmbar-haiku/report.html), also published at https://judgekeeper.github.io/judgekeeper/examples/llmbar-haiku/ (the pattern is `https://<owner>.github.io/judgekeeper/examples/llmbar-haiku/`).
46
+
47
+ New to judge evaluation? The website in [`website/`](website/) explains it step by step, with a setup guide for API keys and a hands-on tutorial. Preview it locally: `python3 -m http.server --directory website 8000`, then open http://localhost:8000.
48
+
49
+ Already have judge verdicts and human labels in a CSV? One command:
50
+
51
+ ```
52
+ judgekeeper check results.csv --judge verdict --human label --out reports/my-judge/
53
+ ```
54
+
55
+ One row per judgment. `id`, `run`, `input`, `output` and `reason` columns are used when present. Verdicts can be `pass`/`fail`, `true`/`false`, `yes`/`no`, `correct`/`incorrect`, `1`/`0` or text starting with `PASS`/`FAIL`; scores need a rule (`--pass-if "score>=0.5"`); other spellings need `--label-map "good=pass,bad=fail"`. judgekeeper never guesses.
56
+
57
+ - No labels yet? `judgekeeper label items.jsonl` opens a local labeling page (keys 1/2, defer, undo, notes) that writes `labels.csv` as you go. [Details](docs/reference.md#label-a-local-labeling-page).
58
+ - Gate from pytest: `pytest --judgekeeper-report reports/my-judge/report.json`, a `judgekeeper_gate` fixture or a `@pytest.mark.judgekeeper` marker. [Details](docs/reference.md#pytest-plugin).
59
+ - Let a coding agent wire it in: [`skills/judgekeeper/SKILL.md`](skills/judgekeeper/SKILL.md) tells Claude Code and other agents how, step by step.
60
+
61
+ > Status: alpha (0.1.0). The validation report, CI gate and GitHub Action, pytest plugin, labeling page, judge migration, drift attribution and the promptfoo, DeepEval, Inspect AI, MLflow and Langfuse readers work. [`docs/reference.md`](docs/reference.md) has every command, flag, exit code and config key; [`CHANGELOG.md`](CHANGELOG.md) what is in 0.1.0.
62
+
63
+ ## Why
64
+
65
+ Teams grade AI outputs with another model, the "judge". Judges disagree with humans more than raw agreement suggests, flip verdicts between identical runs, and change silently when the provider updates the model. judgekeeper measures a judge against a frozen set of human labels, reports TPR and TNR (with 95% intervals) and kappa rather than raw agreement, and tells you, later, whether the judge is still right.
66
+
67
+ The report leads with TPR (how often the judge passes what humans passed) and TNR (how often it fails what humans failed). Either below 0.80 is "not trustworthy as a gate"; 0.80 to 0.90 is "usable with care". It also shows kappa, a confusion matrix, every disagreement with the judge's rationale, the noise floor across repeated runs ("unknown" with one run, never zero), AB/BA position bias for pairwise items, per-slice numbers, and warnings when there are fewer than 60 labels or the classes split worse than 80/20.
68
+
69
+ ## I have a spreadsheet
70
+
71
+ Judge verdicts and human labels already in one table: `judgekeeper check` above. Give it a `run` column (or several rows per id with different runs) to measure run-to-run noise; with one run the noise floor is reported as unknown and `gate` returns `FLAKY`. Without an `id` column, ids are a hash of input and output and the report says so. In a notebook:
72
+
73
+ ```python
74
+ import judgekeeper
75
+ report = judgekeeper.check_table(df, judge="verdict", human="label") # a DataFrame, a list of dicts or a path
76
+ ```
77
+
78
+ No labels yet? Label in the local page or in Excel or Google Sheets, then freeze the labels as an anchor set:
79
+
80
+ ```
81
+ judgekeeper label items.jsonl --out labels.csv # or: judgekeeper template items.jsonl -o labels.csv
82
+ judgekeeper import-labels labels.csv -o anchors.jsonl # checks every label, skips and lists unlabeled rows, freezes
83
+ ```
84
+
85
+ ## I have a judge function
86
+
87
+ Run it 3 times over the anchor set (pairwise items in both AB and BA order) and get a report:
88
+
89
+ ```python
90
+ import judgekeeper
91
+ def my_judge(item): # item: id, input, output (never the human label)
92
+ return call_my_model(item) # bool, "PASS: ...", 0.8, (verdict, reason) or {"verdict": ..., "reason": ...}
93
+ report = judgekeeper.check_judge(my_judge, "anchors.jsonl", runs=3, fingerprint={"model": "my-model"})
94
+ ```
95
+
96
+ Async functions work too. A judge that raises or returns nothing is recorded as an error, excluded from the metrics and counted, never scored as a fail. From the command line, a Python function or a program in any language (one item as JSON on stdin, a verdict on stdout):
97
+
98
+ ```
99
+ judgekeeper judge anchors.jsonl --callable mypkg.judges:my_judge --runs 3 --out runs/mine/
100
+ judgekeeper judge anchors.jsonl --exec "node judge.js" --runs 3 --out runs/mine/
101
+ judgekeeper validate anchors.jsonl runs/mine/ --out reports/mine/
102
+ ```
103
+
104
+ judgekeeper prints the number of judge calls first and asks for `--yes` above 1,000. The `--exec` contract, with 10-line Node and Python judges, is in [`docs/reference.md`](docs/reference.md). Built-in Anthropic and OpenAI-compatible runners (`--runner anthropic|openai`, any OpenAI-compatible endpoint via `--base-url`) are documented there too. Keys stay in environment variables; nothing judgekeeper writes or prints contains one.
105
+
106
+ ## I use a framework
107
+
108
+ Point judgekeeper at the files your eval tool already writes and add human labels. Your eval code does not change.
109
+
110
+ promptfoo (human labels from web-UI ratings, or `--labels`; repeats from `--repeat 3`):
111
+
112
+ ```
113
+ promptfoo eval -o results.json --repeat 3
114
+ judgekeeper import promptfoo results.json --metric helpfulness --out reports/helpfulness/
115
+ ```
116
+
117
+ DeepEval (one `test_run_*.json` per run in a results folder; labels from `--labels`):
118
+
119
+ ```
120
+ export DEEPEVAL_RESULTS_FOLDER=deepeval-results # then run your DeepEval tests 3 times
121
+ judgekeeper import deepeval deepeval-results/ --metric "Correctness [GEval]" --labels labels.csv --out reports/correctness/
122
+ ```
123
+
124
+ Inspect AI (epochs are runs; labels from score edits or `--labels`; `.eval` logs need the `judgekeeper[inspect]` extra):
125
+
126
+ ```
127
+ inspect eval task.py --epochs 3 --log-format json
128
+ judgekeeper import inspect logs/ --metric model_graded_qa --labels labels.csv --out reports/qa/
129
+ ```
130
+
131
+ MLflow (judge and human assessments on your traces; each evaluation run is a run; needs the `judgekeeper[mlflow]` extra):
132
+
133
+ ```
134
+ judgekeeper import mlflow --experiment my-app-eval --metric correctness --out reports/correctness/
135
+ ```
136
+
137
+ Langfuse (judge and human scores from the public API; keys from `LANGFUSE_PUBLIC_KEY` and `LANGFUSE_SECRET_KEY`):
138
+
139
+ ```
140
+ judgekeeper import langfuse --judge-score helpfulness --human-score helpfulness_human --from 2026-09-01 --pass-if "score>=0.5" --out reports/helpfulness/
141
+ ```
142
+
143
+ Each writes the same `report.json` and `report.html` as `check`. With `--anchors-out anchors.jsonl`, the MLflow and Langfuse imports also freeze the labeled items as an anchor set, so you can re-judge them with `judgekeeper judge` for a real noise floor. How to get each file, how labels get in and each tool's traps: [`docs/integrations/`](docs/integrations/). Any other tool can export ScoreRecords (`target_id, name, annotator_kind, label, score, explanation, run, input, output, evaluator, created_at`) and use `judgekeeper import records`, with `--map` for renamed columns; `judgekeeper export records` writes judgekeeper's runs in that format.
144
+
145
+ ## Keep checking
146
+
147
+ ```
148
+ judgekeeper baseline set reports/my-judge/report.json # commit .judgekeeper/baseline.json
149
+ judgekeeper gate reports/my-judge/report.json # PASS, FAIL, FLAKY, JUDGE_CHANGED, ANCHORS_CHANGED
150
+ judgekeeper migrate anchors.jsonl runs/old/ runs/new/ --out reports/migration/
151
+ judgekeeper attribute reports/now/report.json --app-score-before X --app-score-after Y
152
+ ```
153
+
154
+ `gate` never fails on a change inside the noise band. In CI, the GitHub Action (`action.yml`) runs judge, validate and gate, and the pytest plugin gates a report your pipeline already wrote. A judge field that is unknown on either side (imported data rarely records temperature or snapshot) is a warning, not a block, unless you pass `--require-fingerprint`. `migrate` compares an old and a new judge item by item; `attribute` says whether a score moved because your system changed or the judge did. Details: [`docs/reference.md`](docs/reference.md).
155
+
156
+ ## First results
157
+
158
+ The judge is strong on ordinary items and weak on adversarial ones: `claude-haiku-4-5-20251001` reaches kappa 0.94 against human labels on LLMBar's Natural slice but 0.56 on Adversarial/GPTOut ([report](docs/examples/llmbar-haiku/report.html), data in [`docs/examples/llmbar-haiku/`](docs/examples/llmbar-haiku/)).
159
+
160
+ The run: LLMBar (Natural and Adversarial subsets, 419 human-labeled pairwise items) judged by `claude-haiku-4-5-20251001` at temperature 0, 3 runs, AB and BA. `scripts/llmbar_haiku_demo.sh` runs it end to end and writes the report to `docs/examples/llmbar-haiku/`.
161
+
162
+ Run on 2026-10-01 against `api.anthropic.com`. Every number below is quoted from `docs/examples/llmbar-haiku/report.json` (human-readable version: `report.html` in the same folder). TPR and TNR treat human label `A` as the positive class.
163
+
164
+ Verdict: **Usable as a gate: kappa 0.83, TPR 0.95, TNR 0.88 against human labels.**
165
+
166
+ | Run | Kappa | TPR | TNR |
167
+ |---|---|---|---|
168
+ | 1 | 0.82 | 0.95 | 0.87 |
169
+ | 2 | 0.85 | 0.96 | 0.89 |
170
+ | 3 | 0.82 | 0.95 | 0.87 |
171
+ | Mean | 0.83 | 0.95 | 0.88 |
172
+
173
+ By slice (mean over runs):
174
+
175
+ | Slice | Items | Kappa | TPR | TNR |
176
+ |---|---|---|---|---|
177
+ | Natural | 100 | 0.94 | 0.98 | 0.97 |
178
+ | Adversarial/GPTInst | 92 | 0.92 | 1.00 | 0.92 |
179
+ | Adversarial/Neighbor | 134 | 0.84 | 0.96 | 0.88 |
180
+ | Adversarial/Manual | 46 | 0.64 | 0.91 | 0.74 |
181
+ | Adversarial/GPTOut | 47 | 0.56 | 0.86 | 0.71 |
182
+
183
+ - Noise floor: mean pairwise kappa between runs 0.95. 3.6% of items changed verdict in at least one run.
184
+ - Position bias: 9.9% of judgments changed when the two outputs were swapped (AB vs BA). Kappa on the BA order alone was 0.80.
185
+ - The judge's TNR is lower than its TPR. Its errors cluster in the hardest adversarial slices (GPTOut, Manual), where kappa drops to 0.56 and 0.64.
186
+
187
+ ## Why this and not X
188
+
189
+ - Vendor "align evals" features (LangSmith, Arize, MLflow, Ragas, Confident AI) do one-off calibration. judgekeeper covers what happens after: time, change, and gating.
190
+ - RAND's Judge Reliability Harness generates bias probes but has no human-kappa, no time series, no CI, and only supports OpenAI.
191
+ - Several zero-star scripts from Aug to Sep 2026 sketch anchor-set drift checks. judgekeeper aims to be the maintained, multi-provider, framework-integrated version with published numbers.
192
+
193
+ Full landscape and sources: `docs/research/00-judge-reliability-synthesis.md`.
194
+
195
+ ## Roadmap
196
+
197
+ 1. Core validation report on a public human-labeled dataset (done; LLMBar above).
198
+ 2. GitHub Action and noise-floor gate (done).
199
+ 3. Judge migration and drift attribution (built; the live migration demo is pending a key).
200
+ 4. Use without a framework: `check`, `check_table`, `check_judge`, `--callable`, `--exec`, `demo`, spreadsheet labels (done).
201
+ 5. Readers for promptfoo, DeepEval and Inspect AI files, and the ScoreRecord import/export format (done). MLflow and Langfuse readers, with re-judging of imported labels (done).
202
+ 6. Launch readiness: labeling page, pytest plugin, agent skill, release and Pages workflows (done; publishing is the maintainer's manual step, [`docs/RELEASING.md`](docs/RELEASING.md)).
203
+ 7. Demo agent (Claude Agent SDK), clarification (ask-vs-act) judge, MCP server.
204
+
205
+ ## License
206
+
207
+ MIT.