pergamon-bench 0.8.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,6 @@
1
+ """Pergamon Bench grading: rubric judging, weighted legal scoring, rewardkit criteria, `pbench` CLI.
2
+
3
+ Does not need Harbor: the verifier container installs only this package.
4
+ """
5
+
6
+ __version__ = "0.8.0"
pbench_grader/cache.py ADDED
@@ -0,0 +1,69 @@
1
+ """File cache for judge verdicts.
2
+
3
+ Key = sha256(answer, criterion text, judge model, prompt hash, reasoning/temperature).
4
+ Re-running a job or a regrade with the same answer and judge then costs nothing.
5
+ Off unless a directory is configured (``--cache-dir`` for ``pbench grade`` or the
6
+ ``PBENCH_JUDGE_CACHE_DIR`` environment variable).
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import hashlib
12
+ import json
13
+ import os
14
+ from pathlib import Path
15
+ from typing import Any
16
+
17
+
18
+ def cache_dir_from_env() -> Path | None:
19
+ value = os.environ.get("PBENCH_JUDGE_CACHE_DIR")
20
+ return Path(value).expanduser() if value else None
21
+
22
+
23
+ class JudgeCache:
24
+ def __init__(self, directory: Path | None) -> None:
25
+ self.directory = directory
26
+ self.hits = 0
27
+ self.misses = 0
28
+ if directory is not None:
29
+ directory.mkdir(parents=True, exist_ok=True)
30
+
31
+ @property
32
+ def enabled(self) -> bool:
33
+ return self.directory is not None
34
+
35
+ @staticmethod
36
+ def key(*parts: str) -> str:
37
+ h = hashlib.sha256()
38
+ for part in parts:
39
+ h.update(part.encode("utf-8"))
40
+ h.update(b"\0")
41
+ return h.hexdigest()
42
+
43
+ def _path(self, key: str) -> Path:
44
+ assert self.directory is not None
45
+ return self.directory / key[:2] / f"{key}.json"
46
+
47
+ def get(self, key: str) -> dict[str, Any] | None:
48
+ if not self.enabled:
49
+ return None
50
+ path = self._path(key)
51
+ if not path.exists():
52
+ self.misses += 1
53
+ return None
54
+ try:
55
+ data = json.loads(path.read_text(encoding="utf-8"))
56
+ except (OSError, ValueError):
57
+ self.misses += 1
58
+ return None
59
+ self.hits += 1
60
+ return data
61
+
62
+ def put(self, key: str, value: dict[str, Any]) -> None:
63
+ if not self.enabled:
64
+ return
65
+ path = self._path(key)
66
+ path.parent.mkdir(parents=True, exist_ok=True)
67
+ tmp = path.with_suffix(".tmp")
68
+ tmp.write_text(json.dumps(value, ensure_ascii=False), encoding="utf-8")
69
+ tmp.replace(path)
pbench_grader/cli.py ADDED
@@ -0,0 +1,456 @@
1
+ """The grading subcommands of the ``pergamon-bench`` command (no Harbor needed).
2
+
3
+ pergamon-bench grade --tasks ../pergamon-bench/tasks --answers answers.jsonl -o graded/
4
+ pergamon-bench summary jobs/<job-dir> [--markdown]
5
+ pergamon-bench manifest --tasks ../pergamon-bench/tasks
6
+
7
+ ``add_commands`` registers them on a parser; ``run`` executes the parsed arguments.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import argparse
13
+ import asyncio
14
+ import json
15
+ import sys
16
+ from pathlib import Path
17
+ from typing import Any
18
+
19
+ from . import versions
20
+ from .grading import GradingSettings, grade_answer
21
+ from .rk import (
22
+ DEFAULT_ADJUDICATOR_MODEL,
23
+ DEFAULT_JUDGE_REASONING_EFFORT,
24
+ DEFAULT_JUDGE_TEMPERATURE,
25
+ )
26
+ from .rk import DEFAULT_JUDGE_MODELS as DEFAULT_JUDGE
27
+ from .rubric import load_rubric
28
+
29
+ # ── pbench grade ────────────────────────────────────────────────────────
30
+
31
+
32
+ def _read_answers(path: Path) -> list[dict[str, Any]]:
33
+ """JSONL rows: {"task": "<task dir name>", "answer": "..."} or {"task", "answer_path"}."""
34
+ rows: list[dict[str, Any]] = []
35
+ for line_no, line in enumerate(path.read_text(encoding="utf-8").splitlines(), start=1):
36
+ if not line.strip():
37
+ continue
38
+ row = json.loads(line)
39
+ if "task" not in row:
40
+ raise ValueError(f"{path}:{line_no}: missing 'task'")
41
+ if "answer" not in row:
42
+ answer_path = row.get("answer_path")
43
+ if not answer_path:
44
+ raise ValueError(f"{path}:{line_no}: needs 'answer' or 'answer_path'")
45
+ row["answer"] = (path.parent / answer_path).read_text(encoding="utf-8")
46
+ rows.append(row)
47
+ return rows
48
+
49
+
50
+ async def _grade_all(args: argparse.Namespace) -> int:
51
+ settings = GradingSettings(
52
+ judge_models=[m.strip() for m in args.judge_models.split(",") if m.strip()],
53
+ reasoning_effort=args.judge_reasoning_effort,
54
+ temperature=args.judge_temperature,
55
+ concurrency=args.concurrency,
56
+ adjudicator_model=None
57
+ if (args.adjudicator_model or "").lower() == "none"
58
+ else args.adjudicator_model,
59
+ **({"cache_dir": Path(args.cache_dir)} if args.cache_dir else {}),
60
+ )
61
+ rows = _read_answers(Path(args.answers))
62
+ out_root = Path(args.out)
63
+ summary_rows = []
64
+ failures = 0
65
+ for row in rows:
66
+ task_dir = Path(args.tasks) / row["task"]
67
+ rubric = load_rubric(task_dir / "tests" / "rubric.yaml")
68
+ result = await grade_answer(rubric, str(row["answer"]), settings)
69
+ out_dir = out_root / row["task"]
70
+ result.write(out_dir, extra={"answer_source": str(args.answers)})
71
+ (out_dir / "reward.json").write_text(
72
+ json.dumps(result.rewards(), indent=1), encoding="utf-8"
73
+ )
74
+ if result.ungradeable:
75
+ failures += 1
76
+ sc = result.score
77
+ summary_rows.append(
78
+ {
79
+ "task": row["task"],
80
+ "reward": sc.reward,
81
+ "pass": sc.passed,
82
+ "pass_clean": sc.pass_clean,
83
+ "q_final": sc.q_final,
84
+ "q": sc.q,
85
+ "p": sc.p,
86
+ "ungradeable": len(result.ungradeable),
87
+ "judge_cost_usd": round(result.usage.cost_usd, 4),
88
+ }
89
+ )
90
+ print(
91
+ f"{row['task']:32s} reward={sc.reward} score={_fmt(sc.q_final) or 'n/a'} "
92
+ f"(q={sc.q:.3f} p={_fmt(sc.p) or 'n/a'}) clean={sc.pass_clean}"
93
+ )
94
+ (out_root / "summary.json").write_text(json.dumps(summary_rows, indent=1), encoding="utf-8")
95
+ print(
96
+ f"wrote {out_root}/summary.json "
97
+ f"({len(summary_rows)} tasks, {failures} with ungradeable criteria)"
98
+ )
99
+ return 1 if failures else 0
100
+
101
+
102
+ # ── pbench summary ──────────────────────────────────────────────────────
103
+
104
+
105
+ def _trial_dirs(job_dir: Path) -> list[Path]:
106
+ return sorted(p.parent for p in job_dir.glob("*/result.json"))
107
+
108
+
109
+ def _agent_cost(result: dict[str, Any]) -> float:
110
+ return float(((result.get("agent_result") or {}).get("cost_usd")) or 0.0)
111
+
112
+
113
+ def summarise_job(job_dir: Path) -> dict[str, Any]:
114
+ tasks: list[dict[str, Any]] = []
115
+ categories: dict[str, dict[str, int]] = {}
116
+ case_law = {"at_max": 0, "total": 0, "verified_trials": 0, "trials": 0}
117
+ penalty = {"events": 0, "severity": 0.0, "misuse": 0, "not_covered": 0, "rubric_named": 0}
118
+ grading_versions: set[str] = set()
119
+ for trial in _trial_dirs(job_dir):
120
+ result = json.loads((trial / "result.json").read_text(encoding="utf-8"))
121
+ grading_path = trial / "verifier" / "grading.json"
122
+ grading = (
123
+ json.loads(grading_path.read_text(encoding="utf-8")) if grading_path.exists() else {}
124
+ )
125
+ rewards = (result.get("verifier_result") or {}).get("rewards") or {}
126
+ if grading:
127
+ grading_versions.add(grading_label(grading))
128
+ judge_cost = sum(float(u.get("cost_usd", 0)) for u in grading.get("usage", {}).values())
129
+ for cat, counts in grading.get("categories", {}).items():
130
+ entry = categories.setdefault(cat, {"+1": 0, "0": 0, "-1": 0, "total": 0})
131
+ for k in entry:
132
+ entry[k] += counts.get(k, 0)
133
+ cl = [c for c in grading.get("criteria", []) if c.get("grader") == "fr_case_law"]
134
+ if cl:
135
+ case_law["trials"] += 1
136
+ case_law["total"] += len(cl)
137
+ case_law["at_max"] += sum(1 for c in cl if c.get("at_max"))
138
+ if grading.get("groundedness_enabled") and not grading.get("groundedness_error"):
139
+ case_law["verified_trials"] += 1
140
+ pen = grading.get("penalty") or {}
141
+ penalty["events"] += sum(1 for e in pen.get("events", []) if e.get("scored"))
142
+ penalty["severity"] += float(pen.get("total_severity") or 0.0)
143
+ for k in ("misuse", "not_covered", "rubric_named"):
144
+ penalty[k] += int((pen.get("counts") or {}).get(k, 0))
145
+ agent_meta = ((result.get("agent_result") or {}).get("metadata") or {}).get(
146
+ "pbench_agent"
147
+ ) or {}
148
+ tasks.append(
149
+ {
150
+ "trial": trial.name,
151
+ "task": grading.get("task") or trial.name.split("__")[0],
152
+ "agent": (result.get("agent_info") or {}).get("name"),
153
+ "model": ((result.get("agent_info") or {}).get("model_info") or {}).get("name"),
154
+ "reward": rewards.get("reward"),
155
+ "groundedness": _groundedness_status(grading),
156
+ "pass_clean": grading.get("pass_clean"),
157
+ "q_final": grading.get("q_final"),
158
+ "q": grading.get("q"),
159
+ "p": grading.get("p"),
160
+ "integrity": grading.get("integrity"),
161
+ "ungradeable": len(grading.get("ungradeable", [])),
162
+ "exception": (result.get("exception_info") or {}).get("exception_type"),
163
+ "agent_cost_usd": round(_agent_cost(result), 4),
164
+ "judge_cost_usd": round(judge_cost, 4),
165
+ "tool_failures": sum((agent_meta.get("tools_failed") or {}).values()),
166
+ }
167
+ )
168
+ graded = [t for t in tasks if t["reward"] is not None and not t["exception"]]
169
+ # Trials graded without the groundedness check have no penalty: their reward is a
170
+ # more lenient number, reported under its own label and never averaged with verified
171
+ # ones; the penalty-dependent means are n/a as soon as one trial lacks them.
172
+ unverified = [t for t in graded if t["groundedness"] not in ("on", "none")]
173
+ verified = "yes" if not unverified else ("no" if len(unverified) == len(graded) else "partly")
174
+
175
+ def mean(key: str, *, strict: bool = False) -> float | None:
176
+ vals = [float(t[key]) for t in graded if t.get(key) is not None]
177
+ if strict and len(vals) != len(graded):
178
+ return None
179
+ return round(sum(vals) / len(vals), 4) if vals else None
180
+
181
+ return {
182
+ "job": job_dir.name,
183
+ "trials": len(tasks),
184
+ "graded": len(graded),
185
+ "failed": len(tasks) - len(graded),
186
+ "groundedness": verified,
187
+ "reward_label": "reward" if verified == "yes" else "reward (no-groundedness)",
188
+ "pass_rate": mean("reward"),
189
+ "pass_clean_rate": mean("pass_clean", strict=True),
190
+ "mean_score": mean("q_final", strict=True),
191
+ "mean_q": mean("q"),
192
+ "mean_p": mean("p", strict=True),
193
+ "mean_integrity": mean("integrity", strict=True),
194
+ "ungradeable_criteria": sum(t["ungradeable"] for t in tasks),
195
+ "agent_cost_usd": round(sum(t["agent_cost_usd"] for t in tasks), 4),
196
+ "judge_cost_usd": round(sum(t["judge_cost_usd"] for t in tasks), 4),
197
+ "tool_failures": sum(t["tool_failures"] for t in tasks),
198
+ "categories": {
199
+ k: {**v, "at_max_rate": round(v["+1"] / v["total"], 4) if v["total"] else None}
200
+ for k, v in sorted(categories.items())
201
+ },
202
+ "penalty": {**penalty, "severity": round(penalty["severity"], 4)},
203
+ "case_law": _case_law_summary(case_law),
204
+ "grading_versions": sorted(grading_versions),
205
+ "tasks": tasks,
206
+ }
207
+
208
+
209
+ def _groundedness_status(grading: dict[str, Any]) -> str:
210
+ """The trial's groundedness status; derived for grading.json files that predate the field."""
211
+ status = grading.get("groundedness_status")
212
+ if isinstance(status, str):
213
+ return status
214
+ if grading.get("groundedness"):
215
+ return "on"
216
+ if grading.get("groundedness_error"):
217
+ return "error"
218
+ return "off"
219
+
220
+
221
+ def grading_label(grading: dict[str, Any]) -> str:
222
+ """ "1", or "1 + judge_models" when settings differed from version 1."""
223
+ label = str(grading.get("grading_version") or "unrecorded")
224
+ changes = grading.get("grading_changes") or []
225
+ return f"{label} + {', '.join(changes)}" if changes else label
226
+
227
+
228
+ def _case_law_summary(c: dict[str, int]) -> dict[str, Any] | None:
229
+ """Case-law criteria, reported apart: without the groundedness service they are judged by
230
+ the LLM judge alone, which cannot check that a cited decision exists."""
231
+ if not c["total"]:
232
+ return None
233
+ if c["verified_trials"] == c["trials"]:
234
+ verified = "yes"
235
+ elif c["verified_trials"] == 0:
236
+ verified = "no"
237
+ else:
238
+ verified = "partly"
239
+ return {
240
+ "at_max": c["at_max"],
241
+ "total": c["total"],
242
+ "at_max_rate": round(c["at_max"] / c["total"], 4),
243
+ "citations_verified": verified,
244
+ }
245
+
246
+
247
+ def _fmt(v: Any) -> str:
248
+ if v is None:
249
+ return ""
250
+ return f"{v:.3f}" if isinstance(v, float) else str(v)
251
+
252
+
253
+ def _na(v: Any) -> str:
254
+ return "n/a (no groundedness)" if v is None else str(v)
255
+
256
+
257
+ def markdown_summary(summary: dict[str, Any]) -> str:
258
+ lines = [
259
+ f"## {summary['job']}",
260
+ "",
261
+ f"Trials: {summary['trials']} (graded {summary['graded']}, failed {summary['failed']}) · "
262
+ f"{summary['reward_label']} rate **{summary['pass_rate']}** · "
263
+ f"pass_clean {_na(summary['pass_clean_rate'])} · "
264
+ f"mean score {_na(summary['mean_score'])} "
265
+ f"(q {summary['mean_q']}, p {_na(summary['mean_p'])}) · "
266
+ f"citation integrity {_na(summary['mean_integrity'])} · "
267
+ f"ungradeable {summary['ungradeable_criteria']} · "
268
+ f"cost agent ${summary['agent_cost_usd']} + judge ${summary['judge_cost_usd']} · "
269
+ f"tool failures {summary['tool_failures']}",
270
+ "",
271
+ "Grading version: " + ("; ".join(summary.get("grading_versions") or []) or "none"),
272
+ "",
273
+ f"| task | {summary['reward_label']} | clean | score | q | p | integrity "
274
+ "| agent $ | judge $ | note |",
275
+ "|---|:---:|:---:|---:|---:|---:|---:|---:|---:|---|",
276
+ ]
277
+
278
+ def mark(v: Any) -> str:
279
+ return "" if v is None else ("✓" if v else "✗")
280
+
281
+ for t in summary["tasks"]:
282
+ note = t["exception"] or (f"{t['ungradeable']} ungradeable" if t["ungradeable"] else "")
283
+ if t.get("groundedness") not in ("on", "none", None) and not t["exception"]:
284
+ note = f"no groundedness ({t['groundedness']})" + (f"; {note}" if note else "")
285
+ lines.append(
286
+ f"| {t['task']} | {mark(t['reward'])} | {mark(t['pass_clean'])} | "
287
+ f"{_fmt(t['q_final'])} | {_fmt(t['q'])} | {_fmt(t['p'])} | {_fmt(t['integrity'])} | "
288
+ f"{t['agent_cost_usd']} | {t['judge_cost_usd']} | {note} |"
289
+ )
290
+ pen = summary.get("penalty") or {}
291
+ if pen:
292
+ lines += [
293
+ "",
294
+ f"Citation penalty: {pen['events']} scored events, total severity "
295
+ f"{pen['severity']}; recorded only: {pen['misuse']} misuse, "
296
+ f"{pen['not_covered']} outside register coverage, "
297
+ f"{pen['rubric_named']} named by the rubric.",
298
+ ]
299
+ if cl := summary.get("case_law"):
300
+ note = {
301
+ "yes": "citations verified by the groundedness service",
302
+ "no": "citations NOT verified: judged by the LLM judge alone",
303
+ "partly": "citations verified in some trials only",
304
+ }[cl["citations_verified"]]
305
+ lines += [
306
+ "",
307
+ f"Case-law criteria: {cl['at_max']}/{cl['total']} at their best value "
308
+ f"({cl['at_max_rate']}); {note}.",
309
+ ]
310
+ if summary["categories"]:
311
+ lines += [
312
+ "",
313
+ "| category | +1 | 0 | -1 | total | +1 rate |",
314
+ "|---|---:|---:|---:|---:|---:|",
315
+ ]
316
+ for k, v in summary["categories"].items():
317
+ lines.append(
318
+ f"| {k} | {v['+1']} | {v['0']} | {v['-1']} | {v['total']} | {v['at_max_rate']} |"
319
+ )
320
+ return "\n".join(lines)
321
+
322
+
323
+ # ── pbench manifest ─────────────────────────────────────────────────────
324
+
325
+
326
+ def dataset_pin(tasks_dir: Path) -> dict[str, Any]:
327
+ """Name, version and checksum of the dataset, from Harbor's ``dataset.toml`` next to tasks/.
328
+
329
+ The checksum is SHA-256 over the sorted ``name@digest`` lines of the tasks, where each digest
330
+ is Harbor's content hash of the task directory (``harbor add`` keeps them current).
331
+ """
332
+ import hashlib
333
+ import tomllib
334
+
335
+ toml_path = tasks_dir.parent / "dataset.toml"
336
+ if not toml_path.exists():
337
+ return {
338
+ "path": str(tasks_dir),
339
+ "task_count": sum(1 for p in tasks_dir.iterdir() if p.is_dir()),
340
+ }
341
+ data = tomllib.loads(toml_path.read_text(encoding="utf-8"))
342
+ tasks = sorted(f"{t['name']}@{t['digest']}" for t in data.get("tasks", []))
343
+ return {
344
+ "name": data.get("dataset", {}).get("name"),
345
+ "dataset_version": data.get("dataset", {}).get("version"),
346
+ "checksum": hashlib.sha256("\n".join(tasks).encode()).hexdigest(),
347
+ "task_count": len(tasks),
348
+ }
349
+
350
+
351
+ def benchmark_manifest(
352
+ tasks_dir: Path | None, grading_version: str | None = None
353
+ ) -> dict[str, Any]:
354
+ """Everything that pins a benchmark result: code, dataset and grading version.
355
+
356
+ Harbor and the agent package are reported when installed (they are not needed to grade).
357
+ """
358
+ from importlib.metadata import PackageNotFoundError, version
359
+
360
+ from . import __version__
361
+
362
+ def installed(dist: str) -> str | None:
363
+ try:
364
+ return version(dist)
365
+ except PackageNotFoundError:
366
+ return None
367
+
368
+ dataset: dict[str, Any] | None = None
369
+ if tasks_dir is not None:
370
+ dataset = dataset_pin(tasks_dir)
371
+ grading = versions.get(grading_version) if grading_version else versions.current()
372
+ return {
373
+ "benchmark": "pergamon-bench",
374
+ "pbench_grader_version": __version__,
375
+ "pergamon_bench_version": installed("pergamon-bench"),
376
+ "harbor_version": installed("harbor"),
377
+ "dataset": dataset,
378
+ "grading": {**grading.as_dict(), "scoring": "weighted_legal"},
379
+ }
380
+
381
+
382
+ # ── entry point ──────────────────────────────────────────────────────
383
+
384
+
385
+ COMMANDS = ("grade", "manifest", "summary")
386
+
387
+
388
+ def add_commands(sub: Any) -> None:
389
+ """Add grade / manifest / summary to an argparse subparsers object."""
390
+
391
+ g = sub.add_parser(
392
+ "grade", help="grade answers from a JSONL file against task rubrics (no Harbor)"
393
+ )
394
+ g.add_argument(
395
+ "--tasks", required=True, help="directory of Harbor task dirs (the dataset's tasks/)"
396
+ )
397
+ g.add_argument("--answers", required=True, help="JSONL with {task, answer|answer_path} rows")
398
+ g.add_argument("-o", "--out", default="graded", help="output directory")
399
+ g.add_argument("--judge-models", default=DEFAULT_JUDGE, help="comma-separated LiteLLM models")
400
+ g.add_argument("--judge-reasoning-effort", default=DEFAULT_JUDGE_REASONING_EFFORT)
401
+ g.add_argument("--judge-temperature", type=float, default=DEFAULT_JUDGE_TEMPERATURE)
402
+ g.add_argument(
403
+ "--adjudicator-model", default=DEFAULT_ADJUDICATOR_MODEL, help="`none` for no adjudicator"
404
+ )
405
+ g.add_argument("--concurrency", type=int, default=8)
406
+ g.add_argument(
407
+ "--cache-dir", default=None, help="judge cache directory (or PBENCH_JUDGE_CACHE_DIR)"
408
+ )
409
+
410
+ m = sub.add_parser(
411
+ "manifest", help="print the benchmark manifest (code, dataset, grading version)"
412
+ )
413
+ m.add_argument("--tasks", default=None, help="dataset tasks/ directory (reads ../dataset.toml)")
414
+ m.add_argument(
415
+ "--grading-version",
416
+ default=None,
417
+ choices=sorted(versions.load_versions()[1]),
418
+ help="a past grading version (default: the current one)",
419
+ )
420
+
421
+ s = sub.add_parser("summary", help="summarise a Harbor job directory graded by Pergamon Bench")
422
+ s.add_argument("job_dir")
423
+ s.add_argument("--markdown", action="store_true", help="print a markdown table instead of JSON")
424
+ s.add_argument("-o", "--out", default=None, help="also write the JSON summary to this file")
425
+
426
+
427
+ def run(args: argparse.Namespace) -> int:
428
+ """Execute a parsed grade / manifest / summary command."""
429
+ if args.command == "grade":
430
+ return asyncio.run(_grade_all(args))
431
+ if args.command == "manifest":
432
+ tasks = Path(args.tasks) if args.tasks else None
433
+ print(
434
+ json.dumps(
435
+ benchmark_manifest(tasks, args.grading_version), indent=1, ensure_ascii=False
436
+ )
437
+ )
438
+ return 0
439
+ if args.command == "summary":
440
+ summary = summarise_job(Path(args.job_dir))
441
+ if args.out:
442
+ Path(args.out).write_text(json.dumps(summary, indent=1), encoding="utf-8")
443
+ print(markdown_summary(summary) if args.markdown else json.dumps(summary, indent=1))
444
+ return 0
445
+ return 2
446
+
447
+
448
+ def main(argv: list[str] | None = None) -> int:
449
+ """Standalone use: ``python -m pbench_grader.cli <command> ...``."""
450
+ parser = argparse.ArgumentParser(prog="python -m pbench_grader.cli", description=__doc__)
451
+ add_commands(parser.add_subparsers(dest="command", required=True))
452
+ return run(parser.parse_args(argv))
453
+
454
+
455
+ if __name__ == "__main__":
456
+ sys.exit(main())
@@ -0,0 +1,6 @@
1
+ # rewardkit criteria file. The trial's single reward (reward.json `reward`): 1 when every
2
+ # critical criterion is at its best value, else 0. The answer is judged once per trial;
3
+ # see pbench_grader/rk.py. The graded score and the citation penalty are in grading.json.
4
+ from pbench_grader import rk
5
+
6
+ rk.register_pass(rk.tests_dir())
@@ -0,0 +1,8 @@
1
+ # How the criteria files in this directory combine into reward.json (rewardkit).
2
+ # One output only: Harbor computes pass@k when every trial's reward.json has exactly one
3
+ # key whose value is 0 or 1. pass.py is that key; rubric.py contributes nothing to it and
4
+ # exists so reward-details.json lists every criterion.
5
+ [[reward]]
6
+ name = "reward"
7
+ aggregation = "weighted-mean"
8
+ weights = { pass = 1.0, rubric = 0.0 }
@@ -0,0 +1,6 @@
1
+ # rewardkit criteria file. One check per rubric criterion (at its best value), weighted by
2
+ # importance. Weight 0 in reward.toml: it is here for reward-details.json, not the reward.
3
+ # The answer is judged once per trial; see pbench_grader/rk.py.
4
+ from pbench_grader import rk
5
+
6
+ rk.register_rubric(rk.tests_dir())