verifiers 0.2.2.dev25__py3-none-any.whl → 0.2.2.dev26__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/envs/agentic_judge/__init__.py +14 -2
- verifiers/v1/envs/agentic_judge/env.py +260 -87
- {verifiers-0.2.2.dev25.dist-info → verifiers-0.2.2.dev26.dist-info}/METADATA +1 -1
- {verifiers-0.2.2.dev25.dist-info → verifiers-0.2.2.dev26.dist-info}/RECORD +7 -7
- {verifiers-0.2.2.dev25.dist-info → verifiers-0.2.2.dev26.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev25.dist-info → verifiers-0.2.2.dev26.dist-info}/entry_points.txt +0 -0
- {verifiers-0.2.2.dev25.dist-info → verifiers-0.2.2.dev26.dist-info}/licenses/LICENSE +0 -0
|
@@ -1,3 +1,15 @@
|
|
|
1
|
-
from verifiers.v1.envs.agentic_judge.env import
|
|
1
|
+
from verifiers.v1.envs.agentic_judge.env import (
|
|
2
|
+
AgenticJudgeEnv,
|
|
3
|
+
AgenticJudgeEnvConfig,
|
|
4
|
+
Criterion,
|
|
5
|
+
JudgeTaskConfig,
|
|
6
|
+
ScoreConfig,
|
|
7
|
+
)
|
|
2
8
|
|
|
3
|
-
__all__ = [
|
|
9
|
+
__all__ = [
|
|
10
|
+
"AgenticJudgeEnv",
|
|
11
|
+
"AgenticJudgeEnvConfig",
|
|
12
|
+
"Criterion",
|
|
13
|
+
"JudgeTaskConfig",
|
|
14
|
+
"ScoreConfig",
|
|
15
|
+
]
|
|
@@ -1,100 +1,166 @@
|
|
|
1
|
-
"""agentic-judge: a solver plays the task, a
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
runtime onto the judge's trace, and the env's `finalize()` validates it strictly
|
|
13
|
-
onto the solver's trace — a missing, malformed, or off-scale verdict fails loudly instead
|
|
14
|
-
of clamping to full marks.
|
|
1
|
+
"""agentic-judge: a solver plays the task, a judge verifies it in the same box.
|
|
2
|
+
|
|
3
|
+
A reusable env (`--env.id agentic-judge` over any taskset): the box is
|
|
4
|
+
provisioned from the solver's runtime policy, the solver plays the task in it,
|
|
5
|
+
and a code-executing judge then inspects the work as the agent left it, with
|
|
6
|
+
the solver's full trace record uploaded at `/tmp/trace.json`. The judge grades
|
|
7
|
+
rubric criteria (`[env.task]`: policy prompt, criteria file) and writes its
|
|
8
|
+
verdicts to `/tmp/verdict.json`; `finalize()` validates them strictly onto the
|
|
9
|
+
solver's trace — `judge/<name>` metrics plus a weighted-mean `judge` reward,
|
|
10
|
+
composed with the taskset's own rewards via `[env.score]` (judge-only by
|
|
11
|
+
default).
|
|
15
12
|
"""
|
|
16
13
|
|
|
17
14
|
import json
|
|
18
15
|
import math
|
|
16
|
+
import re
|
|
17
|
+
import tomllib
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
from pydantic import field_validator
|
|
19
21
|
|
|
20
22
|
import verifiers.v1 as vf
|
|
21
|
-
from verifiers.v1.
|
|
23
|
+
from verifiers.v1.types import StrictBaseModel
|
|
22
24
|
|
|
23
|
-
TRANSCRIPT_MD = "/tmp/transcript.md"
|
|
24
|
-
TRANSCRIPT_JSON = "/tmp/transcript.json"
|
|
25
25
|
VERDICT_FILE = "/tmp/verdict.json"
|
|
26
|
+
TRACE_FILE = "/tmp/trace.json"
|
|
26
27
|
|
|
27
|
-
GRADE_PROMPT =
|
|
28
|
+
GRADE_PROMPT = """\
|
|
28
29
|
You are grading another agent's attempt at a task. Verify the work EMPIRICALLY:
|
|
29
|
-
reconstruct what the agent did from its
|
|
30
|
-
|
|
31
|
-
can check.
|
|
30
|
+
reconstruct what the agent did from its trace and test it with real execution
|
|
31
|
+
in your sandbox — never take the trace's word for an outcome you can check."""
|
|
32
32
|
|
|
33
|
+
TASK_SECTION = """\
|
|
33
34
|
## The task the agent was given
|
|
34
35
|
|
|
35
|
-
{
|
|
36
|
+
{prompt}"""
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class Criterion(StrictBaseModel):
|
|
40
|
+
"""One rubric criterion — the plugged rubric judge's format, mirrored so the
|
|
41
|
+
same `criteria` files grade both judges."""
|
|
42
|
+
|
|
43
|
+
name: str
|
|
44
|
+
"""Key for the criterion's metric (`judge/<name>`)."""
|
|
45
|
+
text: str
|
|
46
|
+
weight: float = 1.0
|
|
47
|
+
"""The criterion's share of the reward."""
|
|
48
|
+
choices: list[str] = ["no", "yes"]
|
|
49
|
+
"""Allowed answers, ordered **worst → best**: the first scores 0.0, the last 1.0, the rest
|
|
50
|
+
evenly spaced by rank. Default `["no", "yes"]` is a binary check. Needs >= 2, no duplicates."""
|
|
51
|
+
|
|
52
|
+
@field_validator("choices")
|
|
53
|
+
@classmethod
|
|
54
|
+
def _check_choices(cls, v: list[str]) -> list[str]:
|
|
55
|
+
if len(v) < 2:
|
|
56
|
+
raise ValueError(f"`choices` needs at least two options, got {v}")
|
|
57
|
+
if len(set(v)) != len(v):
|
|
58
|
+
raise ValueError(f"`choices` has duplicate options: {v}")
|
|
59
|
+
return v
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
SOLVED = Criterion(
|
|
63
|
+
name="solved",
|
|
64
|
+
text="The task is fully solved: what the task asked for is achieved, and "
|
|
65
|
+
"you verified it with real execution.",
|
|
66
|
+
)
|
|
36
67
|
|
|
68
|
+
|
|
69
|
+
def _verdict_section(criteria: list[Criterion]) -> str:
|
|
70
|
+
listing = "\n".join(
|
|
71
|
+
f"- {c.name}: {c.text} (answer one of, worst to best: {', '.join(c.choices)})"
|
|
72
|
+
for c in criteria
|
|
73
|
+
)
|
|
74
|
+
return f"""\
|
|
37
75
|
## Your verdict
|
|
38
76
|
|
|
77
|
+
Grade the attempt on these criteria:
|
|
78
|
+
|
|
79
|
+
{listing}
|
|
80
|
+
|
|
39
81
|
When you are done verifying, write your verdict as JSON to `{VERDICT_FILE}`:
|
|
40
82
|
|
|
41
|
-
{{{{"
|
|
83
|
+
{{"verdicts": [{{"name": "<criterion name>", "reason": "<one sentence citing \
|
|
84
|
+
what you verified>", "verdict": "<answer>"}}, ...]}}
|
|
42
85
|
|
|
43
|
-
|
|
44
|
-
|
|
86
|
+
with one entry per criterion, using each criterion's exact name. For each, first
|
|
87
|
+
write the one-sentence reason grounded in what you actually verified, then set
|
|
88
|
+
verdict to exactly one of the options listed in parentheses after that
|
|
89
|
+
criterion."""
|
|
45
90
|
|
|
46
91
|
|
|
47
|
-
def
|
|
48
|
-
"""
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
92
|
+
def _render(template: str, **fields: str) -> str:
|
|
93
|
+
"""Substitute documented placeholders in one pass over the original template —
|
|
94
|
+
str.format would crash on any literal brace in a custom prompt, and sequential
|
|
95
|
+
replaces would re-scan substituted values. An unknown placeholder stays as written."""
|
|
96
|
+
pattern = re.compile(r"\{(" + "|".join(map(re.escape, fields)) + r")\}")
|
|
97
|
+
return pattern.sub(lambda m: fields[m.group(1)], template)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
SANDBOX_NOTE = f"""\
|
|
101
|
+
## Your workspace
|
|
102
|
+
|
|
103
|
+
Your sandbox is the SAME box the graded agent worked in, in the state the agent
|
|
104
|
+
left it — its edits (and any scoring side effects) are applied. The agent's raw
|
|
105
|
+
trace record (JSON: messages, tool calls, and its `info` artifacts) is uploaded
|
|
106
|
+
at `{TRACE_FILE}`. The record can be very large — never dump it whole; peek
|
|
107
|
+
selectively (list its keys, then slice out specific fields with python or jq)
|
|
108
|
+
and pull only what you need. It is complete — it may also carry the task's own
|
|
109
|
+
scores/metrics and reference material (a gold answer, a reference solution,
|
|
110
|
+
held-out tests). Those are context, not your standard: recorded scores can be
|
|
111
|
+
wrong and references can be narrower than the task; do not over-index on how a
|
|
112
|
+
reference solves it. Your verdict is what YOU verified by execution."""
|
|
113
|
+
|
|
114
|
+
HINT_SECTION = """\
|
|
115
|
+
## Hints
|
|
116
|
+
|
|
117
|
+
{hint}"""
|
|
61
118
|
|
|
62
119
|
|
|
63
120
|
class JudgeTask(vf.Task):
|
|
64
121
|
"""The judge's verdict task: the solver task's world mirrored onto the minted
|
|
65
|
-
row,
|
|
66
|
-
starts, verdict scraped off the live box after it exits. `NEEDS_CONTAINER`
|
|
122
|
+
row, the trace record written (and any stale verdict removed) before the
|
|
123
|
+
judge starts, verdict scraped off the live box after it exits. `NEEDS_CONTAINER`
|
|
67
124
|
keeps `Agent.run`'s per-task backstop aligned with the judge's declared need."""
|
|
68
125
|
|
|
69
126
|
NEEDS_CONTAINER = True
|
|
70
127
|
|
|
71
128
|
def __init__(self, data: vf.TaskData, files: dict[str, bytes]) -> None:
|
|
72
129
|
super().__init__(data)
|
|
73
|
-
self.
|
|
130
|
+
self.files = files
|
|
74
131
|
|
|
75
132
|
@classmethod
|
|
76
|
-
def from_trace(cls,
|
|
133
|
+
def from_trace(cls, solution: vf.Trace, config: "JudgeTaskConfig") -> "JudgeTask":
|
|
77
134
|
"""Mint the judge's task from the solver's finished trace."""
|
|
78
|
-
|
|
135
|
+
solved = solution.task.data
|
|
136
|
+
files = {TRACE_FILE: json.dumps(solution.to_record()).encode()}
|
|
137
|
+
template = config.build_prompt()
|
|
138
|
+
body = _render(template, prompt=solved.prompt_text)
|
|
139
|
+
if "{prompt}" not in template:
|
|
140
|
+
# A policy that doesn't place the task statement itself still needs it.
|
|
141
|
+
body += "\n\n" + _render(TASK_SECTION, prompt=solved.prompt_text)
|
|
142
|
+
sections = [body, _verdict_section(config.criteria()), SANDBOX_NOTE]
|
|
143
|
+
if (hint := config.build_hint()) is not None:
|
|
144
|
+
sections.insert(1, _render(HINT_SECTION, hint=hint))
|
|
145
|
+
prompt = "\n\n".join(sections)
|
|
79
146
|
return cls(
|
|
80
147
|
vf.TaskData(
|
|
81
|
-
idx=
|
|
82
|
-
prompt=prompt
|
|
83
|
-
image=
|
|
84
|
-
workdir=
|
|
85
|
-
resources=
|
|
148
|
+
idx=solved.idx,
|
|
149
|
+
prompt=prompt,
|
|
150
|
+
image=solved.image,
|
|
151
|
+
workdir=solved.workdir,
|
|
152
|
+
resources=solved.resources,
|
|
86
153
|
),
|
|
87
|
-
files=
|
|
88
|
-
TRANSCRIPT_MD: solution.transcript.encode(),
|
|
89
|
-
TRANSCRIPT_JSON: json.dumps(solution.to_record()).encode(),
|
|
90
|
-
},
|
|
154
|
+
files=files,
|
|
91
155
|
)
|
|
92
156
|
|
|
93
157
|
async def setup(self, trace: vf.Trace, runtime: vf.Runtime) -> None:
|
|
94
|
-
#
|
|
95
|
-
# judge's own
|
|
96
|
-
|
|
97
|
-
|
|
158
|
+
# The solver had this box first: a pre-seeded verdict must never read as
|
|
159
|
+
# the judge's own, and a file (or planted symlink) at an upload path must
|
|
160
|
+
# never survive it — a symlinked TRACE_FILE would redirect the write onto
|
|
161
|
+
# any file the solver chose.
|
|
162
|
+
await runtime.run(["rm", "-f", VERDICT_FILE, *self.files], env={})
|
|
163
|
+
for path, content in self.files.items():
|
|
98
164
|
await runtime.write(path, content)
|
|
99
165
|
|
|
100
166
|
async def finalize(self, trace: vf.Trace, runtime: vf.Runtime) -> None:
|
|
@@ -106,39 +172,125 @@ class JudgeTask(vf.Task):
|
|
|
106
172
|
except Exception as e:
|
|
107
173
|
raise ValueError(
|
|
108
174
|
f"the judge wrote no verdict to {VERDICT_FILE}; its final act must "
|
|
109
|
-
'be writing {"
|
|
175
|
+
'be writing {"verdicts": [{"name", "reason", "verdict"}, ...]} there'
|
|
110
176
|
) from e
|
|
111
177
|
trace.info["verdict"] = json.loads(raw)
|
|
112
178
|
|
|
113
179
|
|
|
180
|
+
class JudgeTaskConfig(vf.BaseConfig):
|
|
181
|
+
"""The judge's minted task: the grading policy and what lands in its box."""
|
|
182
|
+
|
|
183
|
+
prompt: Path | str | None = None
|
|
184
|
+
"""Grading-policy override: inline text, or a policy file (a value ending in
|
|
185
|
+
`.md`/`.txt` is read from disk). Replaces only the policy body — the verdict
|
|
186
|
+
contract and workspace note are always appended, so a custom policy cannot
|
|
187
|
+
break verdict scraping. May reference `{prompt}` (the solver task's prompt);
|
|
188
|
+
if it doesn't, the task statement is appended after the policy."""
|
|
189
|
+
hint: Path | str | None = None
|
|
190
|
+
"""Optional hints injected as their own section (inline text or a `.md`/
|
|
191
|
+
`.txt` file): task-family pointers into the trace or box — e.g. for math,
|
|
192
|
+
where the reference answer lives in the record; for SWE, to diff the repo
|
|
193
|
+
or read `info.patch`."""
|
|
194
|
+
rubric: Path | None = None
|
|
195
|
+
"""Criteria the judge grades against: a `.toml`/`.json` file with a
|
|
196
|
+
`criteria` list — the plugged rubric judge's format, so the same rubric
|
|
197
|
+
files work for both. None grades the single built-in `solved` criterion."""
|
|
198
|
+
|
|
199
|
+
@staticmethod
|
|
200
|
+
def _resolve(value: Path | str) -> str:
|
|
201
|
+
path = Path(value)
|
|
202
|
+
if isinstance(value, Path) or path.suffix in (".md", ".txt"):
|
|
203
|
+
return path.read_text(encoding="utf-8")
|
|
204
|
+
return str(value)
|
|
205
|
+
|
|
206
|
+
def build_prompt(self) -> str:
|
|
207
|
+
if self.prompt is None:
|
|
208
|
+
return GRADE_PROMPT + "\n\n" + TASK_SECTION
|
|
209
|
+
return self._resolve(self.prompt)
|
|
210
|
+
|
|
211
|
+
def build_hint(self) -> str | None:
|
|
212
|
+
return self._resolve(self.hint) if self.hint is not None else None
|
|
213
|
+
|
|
214
|
+
def criteria(self) -> list[Criterion]:
|
|
215
|
+
if self.rubric is None:
|
|
216
|
+
return [SOLVED]
|
|
217
|
+
text = self.rubric.read_text(encoding="utf-8")
|
|
218
|
+
data = (
|
|
219
|
+
tomllib.loads(text)
|
|
220
|
+
if self.rubric.suffix.lower() == ".toml"
|
|
221
|
+
else json.loads(text)
|
|
222
|
+
)
|
|
223
|
+
items = data.get("criteria", []) if isinstance(data, dict) else data
|
|
224
|
+
criteria = [Criterion.model_validate(item) for item in items]
|
|
225
|
+
if not criteria:
|
|
226
|
+
raise ValueError(f"rubric file '{self.rubric}' lists no criteria")
|
|
227
|
+
names = [criterion.name for criterion in criteria]
|
|
228
|
+
if len(set(names)) != len(names):
|
|
229
|
+
raise ValueError(
|
|
230
|
+
f"rubric file '{self.rubric}' has duplicate criterion names"
|
|
231
|
+
)
|
|
232
|
+
if bad := [c.name for c in criteria if not 0 <= c.weight < math.inf]:
|
|
233
|
+
raise ValueError(
|
|
234
|
+
f"rubric '{self.rubric}' has negative or non-finite criterion "
|
|
235
|
+
f"weights: {bad}"
|
|
236
|
+
)
|
|
237
|
+
if sum(criterion.weight for criterion in criteria) <= 0:
|
|
238
|
+
raise ValueError(f"rubric '{self.rubric}' has no positive criterion weight")
|
|
239
|
+
return criteria
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
class ScoreConfig(vf.BaseConfig):
|
|
243
|
+
"""How the judge's verdict composes with the taskset's own rewards on the
|
|
244
|
+
solver's trace. Judge-only by default."""
|
|
245
|
+
|
|
246
|
+
task_weight: float = 0.0
|
|
247
|
+
"""Scale applied to the taskset's own rewards; 1 keeps them next to the verdict."""
|
|
248
|
+
judge_weight: float = 1.0
|
|
249
|
+
"""Weight of the judge's verdict in the solver's reward."""
|
|
250
|
+
|
|
251
|
+
|
|
114
252
|
class AgenticJudgeEnvConfig(vf.EnvConfig):
|
|
115
253
|
solver: vf.AgentConfig = vf.AgentConfig()
|
|
254
|
+
"""The solver agent. It owns the shared box, so its runtime must be a
|
|
255
|
+
container: `--env.solver.runtime.type docker|prime`."""
|
|
116
256
|
judge: vf.AgentConfig = vf.AgentConfig()
|
|
117
|
-
"""The judge agent.
|
|
118
|
-
|
|
257
|
+
"""The judge agent. It plays in the solver's box; its own runtime policy is
|
|
258
|
+
ignored (overwritten with the solver's)."""
|
|
259
|
+
task: JudgeTaskConfig = JudgeTaskConfig()
|
|
260
|
+
score: ScoreConfig = ScoreConfig()
|
|
119
261
|
|
|
120
262
|
|
|
121
263
|
class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
|
|
122
264
|
def __init__(self, config: AgenticJudgeEnvConfig) -> None:
|
|
265
|
+
# The judge plays in the solver's box, so its effective runtime IS the
|
|
266
|
+
# solver's policy — aligning the config keeps the base env's subprocess
|
|
267
|
+
# warning and the runtime stamped on the judge's trace truthful.
|
|
268
|
+
config.judge = config.judge.model_copy(
|
|
269
|
+
update={"runtime": config.solver.runtime}
|
|
270
|
+
)
|
|
123
271
|
super().__init__(config)
|
|
124
|
-
self.
|
|
272
|
+
self._check_agents()
|
|
273
|
+
# A missing policy file or a malformed rubric fails here, not mid-episode.
|
|
274
|
+
config.task.build_prompt()
|
|
275
|
+
config.task.build_hint()
|
|
276
|
+
config.task.criteria()
|
|
125
277
|
|
|
126
|
-
def
|
|
278
|
+
def _check_agents(self) -> None:
|
|
127
279
|
"""The judge executes real code, never on the host — refuse an impossible
|
|
128
|
-
|
|
129
|
-
|
|
280
|
+
pairing at construction, not after burning a full solver run."""
|
|
281
|
+
judge = self._harnesses["judge"]
|
|
282
|
+
if not judge.EXECUTES_CODE:
|
|
130
283
|
raise ValueError(
|
|
131
284
|
"agentic-judge plays a code-executing judge in its own sandbox, but "
|
|
132
|
-
f"harness {
|
|
285
|
+
f"harness {judge.config.id!r} is a tool-less chat loop — a verdict "
|
|
133
286
|
"that needs no execution is a plugged judge "
|
|
134
287
|
"(--env.taskset.task.judges), not an agent."
|
|
135
288
|
)
|
|
136
|
-
if isinstance(
|
|
289
|
+
if isinstance(self.config.solver.runtime, vf.SubprocessConfig):
|
|
137
290
|
raise ValueError(
|
|
138
|
-
"agentic-judge plays its judge in
|
|
139
|
-
"
|
|
140
|
-
"
|
|
141
|
-
"or prime."
|
|
291
|
+
"agentic-judge plays its judge in the solver's box, but the solver "
|
|
292
|
+
"(which provisions it) resolves to the subprocess runtime; use "
|
|
293
|
+
"--env.solver.runtime.type docker or prime"
|
|
142
294
|
)
|
|
143
295
|
|
|
144
296
|
async def setup(self, agents: vf.Agents) -> None:
|
|
@@ -146,29 +298,50 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
|
|
|
146
298
|
agents.judge.trainable = False
|
|
147
299
|
|
|
148
300
|
async def run(self, task: vf.Task, agents: vf.Agents) -> None:
|
|
149
|
-
|
|
150
|
-
|
|
301
|
+
async with agents.solver.provision(task) as box:
|
|
302
|
+
solution = await agents.solver.run(task, runtime=box)
|
|
303
|
+
judge_task = JudgeTask.from_trace(solution, self.config.task)
|
|
304
|
+
await agents.judge.run(judge_task, runtime=box)
|
|
151
305
|
|
|
152
306
|
async def finalize(self, task: vf.Task, episode: vf.Episode) -> None:
|
|
153
|
-
"""Record the scraped verdict on the SOLVER's trace. Strict on scale: an
|
|
154
|
-
off-scale score raises (a judge answering `95` must not clamp to full
|
|
155
|
-
marks), failing the episode rather than scoring the solver wrong."""
|
|
156
307
|
by_agent = {t.agent_name: t for t in episode.traces}
|
|
157
308
|
solution, verdict = by_agent["solver"], by_agent["judge"]
|
|
158
309
|
data = verdict.info.get("verdict")
|
|
159
|
-
if not isinstance(data, dict):
|
|
310
|
+
if not isinstance(data, dict) or not isinstance(data.get("verdicts"), list):
|
|
160
311
|
raise ValueError(
|
|
161
|
-
f"no
|
|
312
|
+
f"no verdicts on the judge's trace (expected {VERDICT_FILE} with a "
|
|
313
|
+
'"verdicts" list)'
|
|
162
314
|
)
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
315
|
+
criteria = self.config.task.criteria()
|
|
316
|
+
by_criterion = {c.name: c for c in criteria}
|
|
317
|
+
answers: dict[str, str] = {}
|
|
318
|
+
for entry in data["verdicts"]:
|
|
319
|
+
if not isinstance(entry, dict):
|
|
320
|
+
raise ValueError(f"verdict entry {entry!r} is not an object")
|
|
321
|
+
name = str(entry.get("name"))
|
|
322
|
+
if name in answers:
|
|
323
|
+
# Contradictory duplicates must not collapse to whichever came last.
|
|
324
|
+
raise ValueError(f"judge answered criterion {name!r} more than once")
|
|
325
|
+
answers[name] = str(entry.get("verdict"))
|
|
326
|
+
if sorted(answers) != sorted(by_criterion):
|
|
170
327
|
raise ValueError(
|
|
171
|
-
f"
|
|
172
|
-
"
|
|
328
|
+
f"judge verdicts name {sorted(answers)}, expected the rubric's "
|
|
329
|
+
f"{sorted(by_criterion)}"
|
|
173
330
|
)
|
|
174
|
-
|
|
331
|
+
scores: dict[str, float] = {}
|
|
332
|
+
for name, answer in answers.items():
|
|
333
|
+
choices = by_criterion[name].choices
|
|
334
|
+
# An off-menu answer is a judge failure, not a zero score.
|
|
335
|
+
if answer not in choices:
|
|
336
|
+
raise ValueError(
|
|
337
|
+
f"judge answered {answer!r} for '{name}', expected one of {choices}"
|
|
338
|
+
)
|
|
339
|
+
scores[name] = choices.index(answer) / (len(choices) - 1)
|
|
340
|
+
for criterion in criteria:
|
|
341
|
+
solution.record_metric(f"judge/{criterion.name}", scores[criterion.name])
|
|
342
|
+
if self.config.score.task_weight != 1.0:
|
|
343
|
+
for name in solution.rewards:
|
|
344
|
+
solution.rewards[name] *= self.config.score.task_weight
|
|
345
|
+
total = sum(criterion.weight for criterion in criteria)
|
|
346
|
+
reward = sum(c.weight * scores[c.name] for c in criteria) / total
|
|
347
|
+
solution.record_reward("judge", reward, weight=self.config.score.judge_weight)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.dev26
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -235,8 +235,8 @@ verifiers/v1/dialects/base.py,sha256=YZQDnasQseDXK-GHLfJLZgyHIIygS80YhD_sTjmUbfA
|
|
|
235
235
|
verifiers/v1/dialects/chat.py,sha256=DFvjmIW86jZUMWuz-hOqzxC6tiiFibiAzNEJ3JkpvjU,13998
|
|
236
236
|
verifiers/v1/dialects/responses.py,sha256=vVgeT7-aWjtfED2IKlDwTATKn_AQZruHNL6mkFTTZTU,13548
|
|
237
237
|
verifiers/v1/envs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
238
|
-
verifiers/v1/envs/agentic_judge/__init__.py,sha256=
|
|
239
|
-
verifiers/v1/envs/agentic_judge/env.py,sha256=
|
|
238
|
+
verifiers/v1/envs/agentic_judge/__init__.py,sha256=X7vQbbbfmJ_j-mJEfBkuB4a8QUPkUYkpG7yp0xYOEmg,279
|
|
239
|
+
verifiers/v1/envs/agentic_judge/env.py,sha256=yd1hqZphIsJD8Z9eDnBU9aGdXmDaAu91QvcyFWaYy8M,15305
|
|
240
240
|
verifiers/v1/envs/best_of_n/__init__.py,sha256=mubASiwXMNIhnQFDccNAc7yZ0OFsOVK8hH8WIx9b2A4,119
|
|
241
241
|
verifiers/v1/envs/best_of_n/env.py,sha256=KEUJBCVqblnXP_lEwm3mjGdaikBMkecz8UjOFPBhxc0,1987
|
|
242
242
|
verifiers/v1/envs/single_agent/__init__.py,sha256=r0edAc_6gtHBhPzeBGVcdHZZqSdCPZnLMje6eVgyz5U,138
|
|
@@ -327,8 +327,8 @@ verifiers/v1/utils/logging.py,sha256=OcMHA6NsYux3oIzjPuI95rDWmFBHNcHDjZeNIhXTX-Y
|
|
|
327
327
|
verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
|
|
328
328
|
verifiers/v1/utils/sampling.py,sha256=JczGzBn6s3wsIrSY2Hy3hE8m6NreNgsNzhSjDikQqX0,1037
|
|
329
329
|
verifiers/v1/utils/version.py,sha256=75ZtI2NHBmlb52KpcKLUpiSXp8q4bASr7uKeXoCKlT8,1582
|
|
330
|
-
verifiers-0.2.2.
|
|
331
|
-
verifiers-0.2.2.
|
|
332
|
-
verifiers-0.2.2.
|
|
333
|
-
verifiers-0.2.2.
|
|
334
|
-
verifiers-0.2.2.
|
|
330
|
+
verifiers-0.2.2.dev26.dist-info/METADATA,sha256=MIaPrTSbtXpw4KSgSjhcqHTnlGiEKPhyLiVTphUTx10,4540
|
|
331
|
+
verifiers-0.2.2.dev26.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
332
|
+
verifiers-0.2.2.dev26.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
|
|
333
|
+
verifiers-0.2.2.dev26.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
334
|
+
verifiers-0.2.2.dev26.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|