pergamon-bench 0.8.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pbench_grader/__init__.py +6 -0
- pbench_grader/cache.py +69 -0
- pbench_grader/cli.py +456 -0
- pbench_grader/criteria/pass.py +6 -0
- pbench_grader/criteria/reward.toml +8 -0
- pbench_grader/criteria/rubric.py +6 -0
- pbench_grader/grading.py +357 -0
- pbench_grader/grading_versions.toml +38 -0
- pbench_grader/groundedness.py +95 -0
- pbench_grader/grounding.py +217 -0
- pbench_grader/judge.py +292 -0
- pbench_grader/judge_prompts.py +142 -0
- pbench_grader/llm.py +203 -0
- pbench_grader/penalty.py +291 -0
- pbench_grader/ratelimit.py +189 -0
- pbench_grader/rk.py +241 -0
- pbench_grader/rubric.py +69 -0
- pbench_grader/scoring.py +136 -0
- pbench_grader/versions.py +416 -0
- pergamon_bench/__init__.py +6 -0
- pergamon_bench/agent.py +353 -0
- pergamon_bench/cli.py +261 -0
- pergamon_bench/dataset/__init__.py +33 -0
- pergamon_bench/dataset/build.py +362 -0
- pergamon_bench/dataset/schema.py +163 -0
- pergamon_bench/endpoint_agent.py +93 -0
- pergamon_bench/official.py +50 -0
- pergamon_bench/templates/env.example +117 -0
- pergamon_bench/templates/pbench.yaml +74 -0
- pergamon_bench-0.8.0.dist-info/METADATA +74 -0
- pergamon_bench-0.8.0.dist-info/RECORD +35 -0
- pergamon_bench-0.8.0.dist-info/WHEEL +4 -0
- pergamon_bench-0.8.0.dist-info/entry_points.txt +2 -0
- pergamon_bench-0.8.0.dist-info/licenses/LICENSE +202 -0
- pergamon_bench-0.8.0.dist-info/licenses/NOTICE +8 -0
pbench_grader/cache.py
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""File cache for judge verdicts.
|
|
2
|
+
|
|
3
|
+
Key = sha256(answer, criterion text, judge model, prompt hash, reasoning/temperature).
|
|
4
|
+
Re-running a job or a regrade with the same answer and judge then costs nothing.
|
|
5
|
+
Off unless a directory is configured (``--cache-dir`` for ``pbench grade`` or the
|
|
6
|
+
``PBENCH_JUDGE_CACHE_DIR`` environment variable).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import hashlib
|
|
12
|
+
import json
|
|
13
|
+
import os
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def cache_dir_from_env() -> Path | None:
|
|
19
|
+
value = os.environ.get("PBENCH_JUDGE_CACHE_DIR")
|
|
20
|
+
return Path(value).expanduser() if value else None
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class JudgeCache:
|
|
24
|
+
def __init__(self, directory: Path | None) -> None:
|
|
25
|
+
self.directory = directory
|
|
26
|
+
self.hits = 0
|
|
27
|
+
self.misses = 0
|
|
28
|
+
if directory is not None:
|
|
29
|
+
directory.mkdir(parents=True, exist_ok=True)
|
|
30
|
+
|
|
31
|
+
@property
|
|
32
|
+
def enabled(self) -> bool:
|
|
33
|
+
return self.directory is not None
|
|
34
|
+
|
|
35
|
+
@staticmethod
|
|
36
|
+
def key(*parts: str) -> str:
|
|
37
|
+
h = hashlib.sha256()
|
|
38
|
+
for part in parts:
|
|
39
|
+
h.update(part.encode("utf-8"))
|
|
40
|
+
h.update(b"\0")
|
|
41
|
+
return h.hexdigest()
|
|
42
|
+
|
|
43
|
+
def _path(self, key: str) -> Path:
|
|
44
|
+
assert self.directory is not None
|
|
45
|
+
return self.directory / key[:2] / f"{key}.json"
|
|
46
|
+
|
|
47
|
+
def get(self, key: str) -> dict[str, Any] | None:
|
|
48
|
+
if not self.enabled:
|
|
49
|
+
return None
|
|
50
|
+
path = self._path(key)
|
|
51
|
+
if not path.exists():
|
|
52
|
+
self.misses += 1
|
|
53
|
+
return None
|
|
54
|
+
try:
|
|
55
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
56
|
+
except (OSError, ValueError):
|
|
57
|
+
self.misses += 1
|
|
58
|
+
return None
|
|
59
|
+
self.hits += 1
|
|
60
|
+
return data
|
|
61
|
+
|
|
62
|
+
def put(self, key: str, value: dict[str, Any]) -> None:
|
|
63
|
+
if not self.enabled:
|
|
64
|
+
return
|
|
65
|
+
path = self._path(key)
|
|
66
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
67
|
+
tmp = path.with_suffix(".tmp")
|
|
68
|
+
tmp.write_text(json.dumps(value, ensure_ascii=False), encoding="utf-8")
|
|
69
|
+
tmp.replace(path)
|
pbench_grader/cli.py
ADDED
|
@@ -0,0 +1,456 @@
|
|
|
1
|
+
"""The grading subcommands of the ``pergamon-bench`` command (no Harbor needed).
|
|
2
|
+
|
|
3
|
+
pergamon-bench grade --tasks ../pergamon-bench/tasks --answers answers.jsonl -o graded/
|
|
4
|
+
pergamon-bench summary jobs/<job-dir> [--markdown]
|
|
5
|
+
pergamon-bench manifest --tasks ../pergamon-bench/tasks
|
|
6
|
+
|
|
7
|
+
``add_commands`` registers them on a parser; ``run`` executes the parsed arguments.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import argparse
|
|
13
|
+
import asyncio
|
|
14
|
+
import json
|
|
15
|
+
import sys
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
from . import versions
|
|
20
|
+
from .grading import GradingSettings, grade_answer
|
|
21
|
+
from .rk import (
|
|
22
|
+
DEFAULT_ADJUDICATOR_MODEL,
|
|
23
|
+
DEFAULT_JUDGE_REASONING_EFFORT,
|
|
24
|
+
DEFAULT_JUDGE_TEMPERATURE,
|
|
25
|
+
)
|
|
26
|
+
from .rk import DEFAULT_JUDGE_MODELS as DEFAULT_JUDGE
|
|
27
|
+
from .rubric import load_rubric
|
|
28
|
+
|
|
29
|
+
# ── pbench grade ────────────────────────────────────────────────────────
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _read_answers(path: Path) -> list[dict[str, Any]]:
|
|
33
|
+
"""JSONL rows: {"task": "<task dir name>", "answer": "..."} or {"task", "answer_path"}."""
|
|
34
|
+
rows: list[dict[str, Any]] = []
|
|
35
|
+
for line_no, line in enumerate(path.read_text(encoding="utf-8").splitlines(), start=1):
|
|
36
|
+
if not line.strip():
|
|
37
|
+
continue
|
|
38
|
+
row = json.loads(line)
|
|
39
|
+
if "task" not in row:
|
|
40
|
+
raise ValueError(f"{path}:{line_no}: missing 'task'")
|
|
41
|
+
if "answer" not in row:
|
|
42
|
+
answer_path = row.get("answer_path")
|
|
43
|
+
if not answer_path:
|
|
44
|
+
raise ValueError(f"{path}:{line_no}: needs 'answer' or 'answer_path'")
|
|
45
|
+
row["answer"] = (path.parent / answer_path).read_text(encoding="utf-8")
|
|
46
|
+
rows.append(row)
|
|
47
|
+
return rows
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
async def _grade_all(args: argparse.Namespace) -> int:
|
|
51
|
+
settings = GradingSettings(
|
|
52
|
+
judge_models=[m.strip() for m in args.judge_models.split(",") if m.strip()],
|
|
53
|
+
reasoning_effort=args.judge_reasoning_effort,
|
|
54
|
+
temperature=args.judge_temperature,
|
|
55
|
+
concurrency=args.concurrency,
|
|
56
|
+
adjudicator_model=None
|
|
57
|
+
if (args.adjudicator_model or "").lower() == "none"
|
|
58
|
+
else args.adjudicator_model,
|
|
59
|
+
**({"cache_dir": Path(args.cache_dir)} if args.cache_dir else {}),
|
|
60
|
+
)
|
|
61
|
+
rows = _read_answers(Path(args.answers))
|
|
62
|
+
out_root = Path(args.out)
|
|
63
|
+
summary_rows = []
|
|
64
|
+
failures = 0
|
|
65
|
+
for row in rows:
|
|
66
|
+
task_dir = Path(args.tasks) / row["task"]
|
|
67
|
+
rubric = load_rubric(task_dir / "tests" / "rubric.yaml")
|
|
68
|
+
result = await grade_answer(rubric, str(row["answer"]), settings)
|
|
69
|
+
out_dir = out_root / row["task"]
|
|
70
|
+
result.write(out_dir, extra={"answer_source": str(args.answers)})
|
|
71
|
+
(out_dir / "reward.json").write_text(
|
|
72
|
+
json.dumps(result.rewards(), indent=1), encoding="utf-8"
|
|
73
|
+
)
|
|
74
|
+
if result.ungradeable:
|
|
75
|
+
failures += 1
|
|
76
|
+
sc = result.score
|
|
77
|
+
summary_rows.append(
|
|
78
|
+
{
|
|
79
|
+
"task": row["task"],
|
|
80
|
+
"reward": sc.reward,
|
|
81
|
+
"pass": sc.passed,
|
|
82
|
+
"pass_clean": sc.pass_clean,
|
|
83
|
+
"q_final": sc.q_final,
|
|
84
|
+
"q": sc.q,
|
|
85
|
+
"p": sc.p,
|
|
86
|
+
"ungradeable": len(result.ungradeable),
|
|
87
|
+
"judge_cost_usd": round(result.usage.cost_usd, 4),
|
|
88
|
+
}
|
|
89
|
+
)
|
|
90
|
+
print(
|
|
91
|
+
f"{row['task']:32s} reward={sc.reward} score={_fmt(sc.q_final) or 'n/a'} "
|
|
92
|
+
f"(q={sc.q:.3f} p={_fmt(sc.p) or 'n/a'}) clean={sc.pass_clean}"
|
|
93
|
+
)
|
|
94
|
+
(out_root / "summary.json").write_text(json.dumps(summary_rows, indent=1), encoding="utf-8")
|
|
95
|
+
print(
|
|
96
|
+
f"wrote {out_root}/summary.json "
|
|
97
|
+
f"({len(summary_rows)} tasks, {failures} with ungradeable criteria)"
|
|
98
|
+
)
|
|
99
|
+
return 1 if failures else 0
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
# ── pbench summary ──────────────────────────────────────────────────────
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _trial_dirs(job_dir: Path) -> list[Path]:
|
|
106
|
+
return sorted(p.parent for p in job_dir.glob("*/result.json"))
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _agent_cost(result: dict[str, Any]) -> float:
|
|
110
|
+
return float(((result.get("agent_result") or {}).get("cost_usd")) or 0.0)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def summarise_job(job_dir: Path) -> dict[str, Any]:
|
|
114
|
+
tasks: list[dict[str, Any]] = []
|
|
115
|
+
categories: dict[str, dict[str, int]] = {}
|
|
116
|
+
case_law = {"at_max": 0, "total": 0, "verified_trials": 0, "trials": 0}
|
|
117
|
+
penalty = {"events": 0, "severity": 0.0, "misuse": 0, "not_covered": 0, "rubric_named": 0}
|
|
118
|
+
grading_versions: set[str] = set()
|
|
119
|
+
for trial in _trial_dirs(job_dir):
|
|
120
|
+
result = json.loads((trial / "result.json").read_text(encoding="utf-8"))
|
|
121
|
+
grading_path = trial / "verifier" / "grading.json"
|
|
122
|
+
grading = (
|
|
123
|
+
json.loads(grading_path.read_text(encoding="utf-8")) if grading_path.exists() else {}
|
|
124
|
+
)
|
|
125
|
+
rewards = (result.get("verifier_result") or {}).get("rewards") or {}
|
|
126
|
+
if grading:
|
|
127
|
+
grading_versions.add(grading_label(grading))
|
|
128
|
+
judge_cost = sum(float(u.get("cost_usd", 0)) for u in grading.get("usage", {}).values())
|
|
129
|
+
for cat, counts in grading.get("categories", {}).items():
|
|
130
|
+
entry = categories.setdefault(cat, {"+1": 0, "0": 0, "-1": 0, "total": 0})
|
|
131
|
+
for k in entry:
|
|
132
|
+
entry[k] += counts.get(k, 0)
|
|
133
|
+
cl = [c for c in grading.get("criteria", []) if c.get("grader") == "fr_case_law"]
|
|
134
|
+
if cl:
|
|
135
|
+
case_law["trials"] += 1
|
|
136
|
+
case_law["total"] += len(cl)
|
|
137
|
+
case_law["at_max"] += sum(1 for c in cl if c.get("at_max"))
|
|
138
|
+
if grading.get("groundedness_enabled") and not grading.get("groundedness_error"):
|
|
139
|
+
case_law["verified_trials"] += 1
|
|
140
|
+
pen = grading.get("penalty") or {}
|
|
141
|
+
penalty["events"] += sum(1 for e in pen.get("events", []) if e.get("scored"))
|
|
142
|
+
penalty["severity"] += float(pen.get("total_severity") or 0.0)
|
|
143
|
+
for k in ("misuse", "not_covered", "rubric_named"):
|
|
144
|
+
penalty[k] += int((pen.get("counts") or {}).get(k, 0))
|
|
145
|
+
agent_meta = ((result.get("agent_result") or {}).get("metadata") or {}).get(
|
|
146
|
+
"pbench_agent"
|
|
147
|
+
) or {}
|
|
148
|
+
tasks.append(
|
|
149
|
+
{
|
|
150
|
+
"trial": trial.name,
|
|
151
|
+
"task": grading.get("task") or trial.name.split("__")[0],
|
|
152
|
+
"agent": (result.get("agent_info") or {}).get("name"),
|
|
153
|
+
"model": ((result.get("agent_info") or {}).get("model_info") or {}).get("name"),
|
|
154
|
+
"reward": rewards.get("reward"),
|
|
155
|
+
"groundedness": _groundedness_status(grading),
|
|
156
|
+
"pass_clean": grading.get("pass_clean"),
|
|
157
|
+
"q_final": grading.get("q_final"),
|
|
158
|
+
"q": grading.get("q"),
|
|
159
|
+
"p": grading.get("p"),
|
|
160
|
+
"integrity": grading.get("integrity"),
|
|
161
|
+
"ungradeable": len(grading.get("ungradeable", [])),
|
|
162
|
+
"exception": (result.get("exception_info") or {}).get("exception_type"),
|
|
163
|
+
"agent_cost_usd": round(_agent_cost(result), 4),
|
|
164
|
+
"judge_cost_usd": round(judge_cost, 4),
|
|
165
|
+
"tool_failures": sum((agent_meta.get("tools_failed") or {}).values()),
|
|
166
|
+
}
|
|
167
|
+
)
|
|
168
|
+
graded = [t for t in tasks if t["reward"] is not None and not t["exception"]]
|
|
169
|
+
# Trials graded without the groundedness check have no penalty: their reward is a
|
|
170
|
+
# more lenient number, reported under its own label and never averaged with verified
|
|
171
|
+
# ones; the penalty-dependent means are n/a as soon as one trial lacks them.
|
|
172
|
+
unverified = [t for t in graded if t["groundedness"] not in ("on", "none")]
|
|
173
|
+
verified = "yes" if not unverified else ("no" if len(unverified) == len(graded) else "partly")
|
|
174
|
+
|
|
175
|
+
def mean(key: str, *, strict: bool = False) -> float | None:
|
|
176
|
+
vals = [float(t[key]) for t in graded if t.get(key) is not None]
|
|
177
|
+
if strict and len(vals) != len(graded):
|
|
178
|
+
return None
|
|
179
|
+
return round(sum(vals) / len(vals), 4) if vals else None
|
|
180
|
+
|
|
181
|
+
return {
|
|
182
|
+
"job": job_dir.name,
|
|
183
|
+
"trials": len(tasks),
|
|
184
|
+
"graded": len(graded),
|
|
185
|
+
"failed": len(tasks) - len(graded),
|
|
186
|
+
"groundedness": verified,
|
|
187
|
+
"reward_label": "reward" if verified == "yes" else "reward (no-groundedness)",
|
|
188
|
+
"pass_rate": mean("reward"),
|
|
189
|
+
"pass_clean_rate": mean("pass_clean", strict=True),
|
|
190
|
+
"mean_score": mean("q_final", strict=True),
|
|
191
|
+
"mean_q": mean("q"),
|
|
192
|
+
"mean_p": mean("p", strict=True),
|
|
193
|
+
"mean_integrity": mean("integrity", strict=True),
|
|
194
|
+
"ungradeable_criteria": sum(t["ungradeable"] for t in tasks),
|
|
195
|
+
"agent_cost_usd": round(sum(t["agent_cost_usd"] for t in tasks), 4),
|
|
196
|
+
"judge_cost_usd": round(sum(t["judge_cost_usd"] for t in tasks), 4),
|
|
197
|
+
"tool_failures": sum(t["tool_failures"] for t in tasks),
|
|
198
|
+
"categories": {
|
|
199
|
+
k: {**v, "at_max_rate": round(v["+1"] / v["total"], 4) if v["total"] else None}
|
|
200
|
+
for k, v in sorted(categories.items())
|
|
201
|
+
},
|
|
202
|
+
"penalty": {**penalty, "severity": round(penalty["severity"], 4)},
|
|
203
|
+
"case_law": _case_law_summary(case_law),
|
|
204
|
+
"grading_versions": sorted(grading_versions),
|
|
205
|
+
"tasks": tasks,
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _groundedness_status(grading: dict[str, Any]) -> str:
|
|
210
|
+
"""The trial's groundedness status; derived for grading.json files that predate the field."""
|
|
211
|
+
status = grading.get("groundedness_status")
|
|
212
|
+
if isinstance(status, str):
|
|
213
|
+
return status
|
|
214
|
+
if grading.get("groundedness"):
|
|
215
|
+
return "on"
|
|
216
|
+
if grading.get("groundedness_error"):
|
|
217
|
+
return "error"
|
|
218
|
+
return "off"
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def grading_label(grading: dict[str, Any]) -> str:
|
|
222
|
+
""" "1", or "1 + judge_models" when settings differed from version 1."""
|
|
223
|
+
label = str(grading.get("grading_version") or "unrecorded")
|
|
224
|
+
changes = grading.get("grading_changes") or []
|
|
225
|
+
return f"{label} + {', '.join(changes)}" if changes else label
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _case_law_summary(c: dict[str, int]) -> dict[str, Any] | None:
|
|
229
|
+
"""Case-law criteria, reported apart: without the groundedness service they are judged by
|
|
230
|
+
the LLM judge alone, which cannot check that a cited decision exists."""
|
|
231
|
+
if not c["total"]:
|
|
232
|
+
return None
|
|
233
|
+
if c["verified_trials"] == c["trials"]:
|
|
234
|
+
verified = "yes"
|
|
235
|
+
elif c["verified_trials"] == 0:
|
|
236
|
+
verified = "no"
|
|
237
|
+
else:
|
|
238
|
+
verified = "partly"
|
|
239
|
+
return {
|
|
240
|
+
"at_max": c["at_max"],
|
|
241
|
+
"total": c["total"],
|
|
242
|
+
"at_max_rate": round(c["at_max"] / c["total"], 4),
|
|
243
|
+
"citations_verified": verified,
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def _fmt(v: Any) -> str:
|
|
248
|
+
if v is None:
|
|
249
|
+
return ""
|
|
250
|
+
return f"{v:.3f}" if isinstance(v, float) else str(v)
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def _na(v: Any) -> str:
|
|
254
|
+
return "n/a (no groundedness)" if v is None else str(v)
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def markdown_summary(summary: dict[str, Any]) -> str:
|
|
258
|
+
lines = [
|
|
259
|
+
f"## {summary['job']}",
|
|
260
|
+
"",
|
|
261
|
+
f"Trials: {summary['trials']} (graded {summary['graded']}, failed {summary['failed']}) · "
|
|
262
|
+
f"{summary['reward_label']} rate **{summary['pass_rate']}** · "
|
|
263
|
+
f"pass_clean {_na(summary['pass_clean_rate'])} · "
|
|
264
|
+
f"mean score {_na(summary['mean_score'])} "
|
|
265
|
+
f"(q {summary['mean_q']}, p {_na(summary['mean_p'])}) · "
|
|
266
|
+
f"citation integrity {_na(summary['mean_integrity'])} · "
|
|
267
|
+
f"ungradeable {summary['ungradeable_criteria']} · "
|
|
268
|
+
f"cost agent ${summary['agent_cost_usd']} + judge ${summary['judge_cost_usd']} · "
|
|
269
|
+
f"tool failures {summary['tool_failures']}",
|
|
270
|
+
"",
|
|
271
|
+
"Grading version: " + ("; ".join(summary.get("grading_versions") or []) or "none"),
|
|
272
|
+
"",
|
|
273
|
+
f"| task | {summary['reward_label']} | clean | score | q | p | integrity "
|
|
274
|
+
"| agent $ | judge $ | note |",
|
|
275
|
+
"|---|:---:|:---:|---:|---:|---:|---:|---:|---:|---|",
|
|
276
|
+
]
|
|
277
|
+
|
|
278
|
+
def mark(v: Any) -> str:
|
|
279
|
+
return "" if v is None else ("✓" if v else "✗")
|
|
280
|
+
|
|
281
|
+
for t in summary["tasks"]:
|
|
282
|
+
note = t["exception"] or (f"{t['ungradeable']} ungradeable" if t["ungradeable"] else "")
|
|
283
|
+
if t.get("groundedness") not in ("on", "none", None) and not t["exception"]:
|
|
284
|
+
note = f"no groundedness ({t['groundedness']})" + (f"; {note}" if note else "")
|
|
285
|
+
lines.append(
|
|
286
|
+
f"| {t['task']} | {mark(t['reward'])} | {mark(t['pass_clean'])} | "
|
|
287
|
+
f"{_fmt(t['q_final'])} | {_fmt(t['q'])} | {_fmt(t['p'])} | {_fmt(t['integrity'])} | "
|
|
288
|
+
f"{t['agent_cost_usd']} | {t['judge_cost_usd']} | {note} |"
|
|
289
|
+
)
|
|
290
|
+
pen = summary.get("penalty") or {}
|
|
291
|
+
if pen:
|
|
292
|
+
lines += [
|
|
293
|
+
"",
|
|
294
|
+
f"Citation penalty: {pen['events']} scored events, total severity "
|
|
295
|
+
f"{pen['severity']}; recorded only: {pen['misuse']} misuse, "
|
|
296
|
+
f"{pen['not_covered']} outside register coverage, "
|
|
297
|
+
f"{pen['rubric_named']} named by the rubric.",
|
|
298
|
+
]
|
|
299
|
+
if cl := summary.get("case_law"):
|
|
300
|
+
note = {
|
|
301
|
+
"yes": "citations verified by the groundedness service",
|
|
302
|
+
"no": "citations NOT verified: judged by the LLM judge alone",
|
|
303
|
+
"partly": "citations verified in some trials only",
|
|
304
|
+
}[cl["citations_verified"]]
|
|
305
|
+
lines += [
|
|
306
|
+
"",
|
|
307
|
+
f"Case-law criteria: {cl['at_max']}/{cl['total']} at their best value "
|
|
308
|
+
f"({cl['at_max_rate']}); {note}.",
|
|
309
|
+
]
|
|
310
|
+
if summary["categories"]:
|
|
311
|
+
lines += [
|
|
312
|
+
"",
|
|
313
|
+
"| category | +1 | 0 | -1 | total | +1 rate |",
|
|
314
|
+
"|---|---:|---:|---:|---:|---:|",
|
|
315
|
+
]
|
|
316
|
+
for k, v in summary["categories"].items():
|
|
317
|
+
lines.append(
|
|
318
|
+
f"| {k} | {v['+1']} | {v['0']} | {v['-1']} | {v['total']} | {v['at_max_rate']} |"
|
|
319
|
+
)
|
|
320
|
+
return "\n".join(lines)
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
# ── pbench manifest ─────────────────────────────────────────────────────
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def dataset_pin(tasks_dir: Path) -> dict[str, Any]:
|
|
327
|
+
"""Name, version and checksum of the dataset, from Harbor's ``dataset.toml`` next to tasks/.
|
|
328
|
+
|
|
329
|
+
The checksum is SHA-256 over the sorted ``name@digest`` lines of the tasks, where each digest
|
|
330
|
+
is Harbor's content hash of the task directory (``harbor add`` keeps them current).
|
|
331
|
+
"""
|
|
332
|
+
import hashlib
|
|
333
|
+
import tomllib
|
|
334
|
+
|
|
335
|
+
toml_path = tasks_dir.parent / "dataset.toml"
|
|
336
|
+
if not toml_path.exists():
|
|
337
|
+
return {
|
|
338
|
+
"path": str(tasks_dir),
|
|
339
|
+
"task_count": sum(1 for p in tasks_dir.iterdir() if p.is_dir()),
|
|
340
|
+
}
|
|
341
|
+
data = tomllib.loads(toml_path.read_text(encoding="utf-8"))
|
|
342
|
+
tasks = sorted(f"{t['name']}@{t['digest']}" for t in data.get("tasks", []))
|
|
343
|
+
return {
|
|
344
|
+
"name": data.get("dataset", {}).get("name"),
|
|
345
|
+
"dataset_version": data.get("dataset", {}).get("version"),
|
|
346
|
+
"checksum": hashlib.sha256("\n".join(tasks).encode()).hexdigest(),
|
|
347
|
+
"task_count": len(tasks),
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def benchmark_manifest(
|
|
352
|
+
tasks_dir: Path | None, grading_version: str | None = None
|
|
353
|
+
) -> dict[str, Any]:
|
|
354
|
+
"""Everything that pins a benchmark result: code, dataset and grading version.
|
|
355
|
+
|
|
356
|
+
Harbor and the agent package are reported when installed (they are not needed to grade).
|
|
357
|
+
"""
|
|
358
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
359
|
+
|
|
360
|
+
from . import __version__
|
|
361
|
+
|
|
362
|
+
def installed(dist: str) -> str | None:
|
|
363
|
+
try:
|
|
364
|
+
return version(dist)
|
|
365
|
+
except PackageNotFoundError:
|
|
366
|
+
return None
|
|
367
|
+
|
|
368
|
+
dataset: dict[str, Any] | None = None
|
|
369
|
+
if tasks_dir is not None:
|
|
370
|
+
dataset = dataset_pin(tasks_dir)
|
|
371
|
+
grading = versions.get(grading_version) if grading_version else versions.current()
|
|
372
|
+
return {
|
|
373
|
+
"benchmark": "pergamon-bench",
|
|
374
|
+
"pbench_grader_version": __version__,
|
|
375
|
+
"pergamon_bench_version": installed("pergamon-bench"),
|
|
376
|
+
"harbor_version": installed("harbor"),
|
|
377
|
+
"dataset": dataset,
|
|
378
|
+
"grading": {**grading.as_dict(), "scoring": "weighted_legal"},
|
|
379
|
+
}
|
|
380
|
+
|
|
381
|
+
|
|
382
|
+
# ── entry point ──────────────────────────────────────────────────────
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
COMMANDS = ("grade", "manifest", "summary")
|
|
386
|
+
|
|
387
|
+
|
|
388
|
+
def add_commands(sub: Any) -> None:
|
|
389
|
+
"""Add grade / manifest / summary to an argparse subparsers object."""
|
|
390
|
+
|
|
391
|
+
g = sub.add_parser(
|
|
392
|
+
"grade", help="grade answers from a JSONL file against task rubrics (no Harbor)"
|
|
393
|
+
)
|
|
394
|
+
g.add_argument(
|
|
395
|
+
"--tasks", required=True, help="directory of Harbor task dirs (the dataset's tasks/)"
|
|
396
|
+
)
|
|
397
|
+
g.add_argument("--answers", required=True, help="JSONL with {task, answer|answer_path} rows")
|
|
398
|
+
g.add_argument("-o", "--out", default="graded", help="output directory")
|
|
399
|
+
g.add_argument("--judge-models", default=DEFAULT_JUDGE, help="comma-separated LiteLLM models")
|
|
400
|
+
g.add_argument("--judge-reasoning-effort", default=DEFAULT_JUDGE_REASONING_EFFORT)
|
|
401
|
+
g.add_argument("--judge-temperature", type=float, default=DEFAULT_JUDGE_TEMPERATURE)
|
|
402
|
+
g.add_argument(
|
|
403
|
+
"--adjudicator-model", default=DEFAULT_ADJUDICATOR_MODEL, help="`none` for no adjudicator"
|
|
404
|
+
)
|
|
405
|
+
g.add_argument("--concurrency", type=int, default=8)
|
|
406
|
+
g.add_argument(
|
|
407
|
+
"--cache-dir", default=None, help="judge cache directory (or PBENCH_JUDGE_CACHE_DIR)"
|
|
408
|
+
)
|
|
409
|
+
|
|
410
|
+
m = sub.add_parser(
|
|
411
|
+
"manifest", help="print the benchmark manifest (code, dataset, grading version)"
|
|
412
|
+
)
|
|
413
|
+
m.add_argument("--tasks", default=None, help="dataset tasks/ directory (reads ../dataset.toml)")
|
|
414
|
+
m.add_argument(
|
|
415
|
+
"--grading-version",
|
|
416
|
+
default=None,
|
|
417
|
+
choices=sorted(versions.load_versions()[1]),
|
|
418
|
+
help="a past grading version (default: the current one)",
|
|
419
|
+
)
|
|
420
|
+
|
|
421
|
+
s = sub.add_parser("summary", help="summarise a Harbor job directory graded by Pergamon Bench")
|
|
422
|
+
s.add_argument("job_dir")
|
|
423
|
+
s.add_argument("--markdown", action="store_true", help="print a markdown table instead of JSON")
|
|
424
|
+
s.add_argument("-o", "--out", default=None, help="also write the JSON summary to this file")
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
def run(args: argparse.Namespace) -> int:
|
|
428
|
+
"""Execute a parsed grade / manifest / summary command."""
|
|
429
|
+
if args.command == "grade":
|
|
430
|
+
return asyncio.run(_grade_all(args))
|
|
431
|
+
if args.command == "manifest":
|
|
432
|
+
tasks = Path(args.tasks) if args.tasks else None
|
|
433
|
+
print(
|
|
434
|
+
json.dumps(
|
|
435
|
+
benchmark_manifest(tasks, args.grading_version), indent=1, ensure_ascii=False
|
|
436
|
+
)
|
|
437
|
+
)
|
|
438
|
+
return 0
|
|
439
|
+
if args.command == "summary":
|
|
440
|
+
summary = summarise_job(Path(args.job_dir))
|
|
441
|
+
if args.out:
|
|
442
|
+
Path(args.out).write_text(json.dumps(summary, indent=1), encoding="utf-8")
|
|
443
|
+
print(markdown_summary(summary) if args.markdown else json.dumps(summary, indent=1))
|
|
444
|
+
return 0
|
|
445
|
+
return 2
|
|
446
|
+
|
|
447
|
+
|
|
448
|
+
def main(argv: list[str] | None = None) -> int:
|
|
449
|
+
"""Standalone use: ``python -m pbench_grader.cli <command> ...``."""
|
|
450
|
+
parser = argparse.ArgumentParser(prog="python -m pbench_grader.cli", description=__doc__)
|
|
451
|
+
add_commands(parser.add_subparsers(dest="command", required=True))
|
|
452
|
+
return run(parser.parse_args(argv))
|
|
453
|
+
|
|
454
|
+
|
|
455
|
+
if __name__ == "__main__":
|
|
456
|
+
sys.exit(main())
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
# rewardkit criteria file. The trial's single reward (reward.json `reward`): 1 when every
|
|
2
|
+
# critical criterion is at its best value, else 0. The answer is judged once per trial;
|
|
3
|
+
# see pbench_grader/rk.py. The graded score and the citation penalty are in grading.json.
|
|
4
|
+
from pbench_grader import rk
|
|
5
|
+
|
|
6
|
+
rk.register_pass(rk.tests_dir())
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
# How the criteria files in this directory combine into reward.json (rewardkit).
|
|
2
|
+
# One output only: Harbor computes pass@k when every trial's reward.json has exactly one
|
|
3
|
+
# key whose value is 0 or 1. pass.py is that key; rubric.py contributes nothing to it and
|
|
4
|
+
# exists so reward-details.json lists every criterion.
|
|
5
|
+
[[reward]]
|
|
6
|
+
name = "reward"
|
|
7
|
+
aggregation = "weighted-mean"
|
|
8
|
+
weights = { pass = 1.0, rubric = 0.0 }
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
# rewardkit criteria file. One check per rubric criterion (at its best value), weighted by
|
|
2
|
+
# importance. Weight 0 in reward.toml: it is here for reward-details.json, not the reward.
|
|
3
|
+
# The answer is judged once per trial; see pbench_grader/rk.py.
|
|
4
|
+
from pbench_grader import rk
|
|
5
|
+
|
|
6
|
+
rk.register_rubric(rk.tests_dir())
|