verifiers 0.2.2.dev25__py3-none-any.whl → 0.2.2.dev27__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/__init__.py +2 -0
- verifiers/v1/cli/dashboard/eval.py +9 -3
- verifiers/v1/envs/agentic_judge/__init__.py +14 -2
- verifiers/v1/envs/agentic_judge/env.py +260 -87
- verifiers/v1/legacy.py +2 -1
- verifiers/v1/push.py +6 -4
- verifiers/v1/trace.py +23 -7
- {verifiers-0.2.2.dev25.dist-info → verifiers-0.2.2.dev27.dist-info}/METADATA +1 -1
- {verifiers-0.2.2.dev25.dist-info → verifiers-0.2.2.dev27.dist-info}/RECORD +12 -12
- {verifiers-0.2.2.dev25.dist-info → verifiers-0.2.2.dev27.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev25.dist-info → verifiers-0.2.2.dev27.dist-info}/entry_points.txt +0 -0
- {verifiers-0.2.2.dev25.dist-info → verifiers-0.2.2.dev27.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/__init__.py
CHANGED
|
@@ -107,6 +107,7 @@ from verifiers.v1.trace import (
|
|
|
107
107
|
EvalRunInfo,
|
|
108
108
|
GenerationSpan,
|
|
109
109
|
ModelCall,
|
|
110
|
+
Reward,
|
|
110
111
|
RunInfo,
|
|
111
112
|
TimeSpan,
|
|
112
113
|
TimeSplit,
|
|
@@ -171,6 +172,7 @@ __all__ = [
|
|
|
171
172
|
"Trace",
|
|
172
173
|
"TraceTask",
|
|
173
174
|
"WireTrace",
|
|
175
|
+
"Reward",
|
|
174
176
|
"Episode",
|
|
175
177
|
"WireEpisode",
|
|
176
178
|
"TRACE_VERSION",
|
|
@@ -341,13 +341,19 @@ def _score_segments(traces: list[Trace], source: str) -> str | None:
|
|
|
341
341
|
return None
|
|
342
342
|
segments = []
|
|
343
343
|
for name in names:
|
|
344
|
-
mean = format_mean(
|
|
345
|
-
traces, lambda t, n=name, s=source: getattr(t, s).get(n, 0.0)
|
|
346
|
-
)
|
|
344
|
+
mean = format_mean(traces, lambda t, n=name, s=source: _score(t, s, n))
|
|
347
345
|
segments.append(f"{name} {mean}")
|
|
348
346
|
return " · ".join(segments)
|
|
349
347
|
|
|
350
348
|
|
|
349
|
+
def _score(trace: Trace, source: str, name: str) -> float:
|
|
350
|
+
"""Rewards carry raw score + weight; the breakdown shows the raw score."""
|
|
351
|
+
if source == "rewards":
|
|
352
|
+
reward = trace.rewards.get(name)
|
|
353
|
+
return reward.score if reward is not None else 0.0
|
|
354
|
+
return trace.metrics.get(name, 0.0)
|
|
355
|
+
|
|
356
|
+
|
|
351
357
|
def _breakdown(scored: list[Trace], done: list[Trace]) -> Table | None:
|
|
352
358
|
"""Score rows read the policy view (`scored` — trainable traces); with several
|
|
353
359
|
roles in play they split per role, each role averaging over its OWN traces (no
|
|
@@ -1,3 +1,15 @@
|
|
|
1
|
-
from verifiers.v1.envs.agentic_judge.env import
|
|
1
|
+
from verifiers.v1.envs.agentic_judge.env import (
|
|
2
|
+
AgenticJudgeEnv,
|
|
3
|
+
AgenticJudgeEnvConfig,
|
|
4
|
+
Criterion,
|
|
5
|
+
JudgeTaskConfig,
|
|
6
|
+
ScoreConfig,
|
|
7
|
+
)
|
|
2
8
|
|
|
3
|
-
__all__ = [
|
|
9
|
+
__all__ = [
|
|
10
|
+
"AgenticJudgeEnv",
|
|
11
|
+
"AgenticJudgeEnvConfig",
|
|
12
|
+
"Criterion",
|
|
13
|
+
"JudgeTaskConfig",
|
|
14
|
+
"ScoreConfig",
|
|
15
|
+
]
|
|
@@ -1,100 +1,166 @@
|
|
|
1
|
-
"""agentic-judge: a solver plays the task, a
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
runtime onto the judge's trace, and the env's `finalize()` validates it strictly
|
|
13
|
-
onto the solver's trace — a missing, malformed, or off-scale verdict fails loudly instead
|
|
14
|
-
of clamping to full marks.
|
|
1
|
+
"""agentic-judge: a solver plays the task, a judge verifies it in the same box.
|
|
2
|
+
|
|
3
|
+
A reusable env (`--env.id agentic-judge` over any taskset): the box is
|
|
4
|
+
provisioned from the solver's runtime policy, the solver plays the task in it,
|
|
5
|
+
and a code-executing judge then inspects the work as the agent left it, with
|
|
6
|
+
the solver's full trace record uploaded at `/tmp/trace.json`. The judge grades
|
|
7
|
+
rubric criteria (`[env.task]`: policy prompt, criteria file) and writes its
|
|
8
|
+
verdicts to `/tmp/verdict.json`; `finalize()` validates them strictly onto the
|
|
9
|
+
solver's trace — `judge/<name>` metrics plus a weighted-mean `judge` reward,
|
|
10
|
+
composed with the taskset's own rewards via `[env.score]` (judge-only by
|
|
11
|
+
default).
|
|
15
12
|
"""
|
|
16
13
|
|
|
17
14
|
import json
|
|
18
15
|
import math
|
|
16
|
+
import re
|
|
17
|
+
import tomllib
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
from pydantic import field_validator
|
|
19
21
|
|
|
20
22
|
import verifiers.v1 as vf
|
|
21
|
-
from verifiers.v1.
|
|
23
|
+
from verifiers.v1.types import StrictBaseModel
|
|
22
24
|
|
|
23
|
-
TRANSCRIPT_MD = "/tmp/transcript.md"
|
|
24
|
-
TRANSCRIPT_JSON = "/tmp/transcript.json"
|
|
25
25
|
VERDICT_FILE = "/tmp/verdict.json"
|
|
26
|
+
TRACE_FILE = "/tmp/trace.json"
|
|
26
27
|
|
|
27
|
-
GRADE_PROMPT =
|
|
28
|
+
GRADE_PROMPT = """\
|
|
28
29
|
You are grading another agent's attempt at a task. Verify the work EMPIRICALLY:
|
|
29
|
-
reconstruct what the agent did from its
|
|
30
|
-
|
|
31
|
-
can check.
|
|
30
|
+
reconstruct what the agent did from its trace and test it with real execution
|
|
31
|
+
in your sandbox — never take the trace's word for an outcome you can check."""
|
|
32
32
|
|
|
33
|
+
TASK_SECTION = """\
|
|
33
34
|
## The task the agent was given
|
|
34
35
|
|
|
35
|
-
{
|
|
36
|
+
{prompt}"""
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class Criterion(StrictBaseModel):
|
|
40
|
+
"""One rubric criterion — the plugged rubric judge's format, mirrored so the
|
|
41
|
+
same `criteria` files grade both judges."""
|
|
42
|
+
|
|
43
|
+
name: str
|
|
44
|
+
"""Key for the criterion's metric (`judge/<name>`)."""
|
|
45
|
+
text: str
|
|
46
|
+
weight: float = 1.0
|
|
47
|
+
"""The criterion's share of the reward."""
|
|
48
|
+
choices: list[str] = ["no", "yes"]
|
|
49
|
+
"""Allowed answers, ordered **worst → best**: the first scores 0.0, the last 1.0, the rest
|
|
50
|
+
evenly spaced by rank. Default `["no", "yes"]` is a binary check. Needs >= 2, no duplicates."""
|
|
51
|
+
|
|
52
|
+
@field_validator("choices")
|
|
53
|
+
@classmethod
|
|
54
|
+
def _check_choices(cls, v: list[str]) -> list[str]:
|
|
55
|
+
if len(v) < 2:
|
|
56
|
+
raise ValueError(f"`choices` needs at least two options, got {v}")
|
|
57
|
+
if len(set(v)) != len(v):
|
|
58
|
+
raise ValueError(f"`choices` has duplicate options: {v}")
|
|
59
|
+
return v
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
SOLVED = Criterion(
|
|
63
|
+
name="solved",
|
|
64
|
+
text="The task is fully solved: what the task asked for is achieved, and "
|
|
65
|
+
"you verified it with real execution.",
|
|
66
|
+
)
|
|
36
67
|
|
|
68
|
+
|
|
69
|
+
def _verdict_section(criteria: list[Criterion]) -> str:
|
|
70
|
+
listing = "\n".join(
|
|
71
|
+
f"- {c.name}: {c.text} (answer one of, worst to best: {', '.join(c.choices)})"
|
|
72
|
+
for c in criteria
|
|
73
|
+
)
|
|
74
|
+
return f"""\
|
|
37
75
|
## Your verdict
|
|
38
76
|
|
|
77
|
+
Grade the attempt on these criteria:
|
|
78
|
+
|
|
79
|
+
{listing}
|
|
80
|
+
|
|
39
81
|
When you are done verifying, write your verdict as JSON to `{VERDICT_FILE}`:
|
|
40
82
|
|
|
41
|
-
{{{{"
|
|
83
|
+
{{"verdicts": [{{"name": "<criterion name>", "reason": "<one sentence citing \
|
|
84
|
+
what you verified>", "verdict": "<answer>"}}, ...]}}
|
|
42
85
|
|
|
43
|
-
|
|
44
|
-
|
|
86
|
+
with one entry per criterion, using each criterion's exact name. For each, first
|
|
87
|
+
write the one-sentence reason grounded in what you actually verified, then set
|
|
88
|
+
verdict to exactly one of the options listed in parentheses after that
|
|
89
|
+
criterion."""
|
|
45
90
|
|
|
46
91
|
|
|
47
|
-
def
|
|
48
|
-
"""
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
92
|
+
def _render(template: str, **fields: str) -> str:
|
|
93
|
+
"""Substitute documented placeholders in one pass over the original template —
|
|
94
|
+
str.format would crash on any literal brace in a custom prompt, and sequential
|
|
95
|
+
replaces would re-scan substituted values. An unknown placeholder stays as written."""
|
|
96
|
+
pattern = re.compile(r"\{(" + "|".join(map(re.escape, fields)) + r")\}")
|
|
97
|
+
return pattern.sub(lambda m: fields[m.group(1)], template)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
SANDBOX_NOTE = f"""\
|
|
101
|
+
## Your workspace
|
|
102
|
+
|
|
103
|
+
Your sandbox is the SAME box the graded agent worked in, in the state the agent
|
|
104
|
+
left it — its edits (and any scoring side effects) are applied. The agent's raw
|
|
105
|
+
trace record (JSON: messages, tool calls, and its `info` artifacts) is uploaded
|
|
106
|
+
at `{TRACE_FILE}`. The record can be very large — never dump it whole; peek
|
|
107
|
+
selectively (list its keys, then slice out specific fields with python or jq)
|
|
108
|
+
and pull only what you need. It is complete — it may also carry the task's own
|
|
109
|
+
scores/metrics and reference material (a gold answer, a reference solution,
|
|
110
|
+
held-out tests). Those are context, not your standard: recorded scores can be
|
|
111
|
+
wrong and references can be narrower than the task; do not over-index on how a
|
|
112
|
+
reference solves it. Your verdict is what YOU verified by execution."""
|
|
113
|
+
|
|
114
|
+
HINT_SECTION = """\
|
|
115
|
+
## Hints
|
|
116
|
+
|
|
117
|
+
{hint}"""
|
|
61
118
|
|
|
62
119
|
|
|
63
120
|
class JudgeTask(vf.Task):
|
|
64
121
|
"""The judge's verdict task: the solver task's world mirrored onto the minted
|
|
65
|
-
row,
|
|
66
|
-
starts, verdict scraped off the live box after it exits. `NEEDS_CONTAINER`
|
|
122
|
+
row, the trace record written (and any stale verdict removed) before the
|
|
123
|
+
judge starts, verdict scraped off the live box after it exits. `NEEDS_CONTAINER`
|
|
67
124
|
keeps `Agent.run`'s per-task backstop aligned with the judge's declared need."""
|
|
68
125
|
|
|
69
126
|
NEEDS_CONTAINER = True
|
|
70
127
|
|
|
71
128
|
def __init__(self, data: vf.TaskData, files: dict[str, bytes]) -> None:
|
|
72
129
|
super().__init__(data)
|
|
73
|
-
self.
|
|
130
|
+
self.files = files
|
|
74
131
|
|
|
75
132
|
@classmethod
|
|
76
|
-
def from_trace(cls,
|
|
133
|
+
def from_trace(cls, solution: vf.Trace, config: "JudgeTaskConfig") -> "JudgeTask":
|
|
77
134
|
"""Mint the judge's task from the solver's finished trace."""
|
|
78
|
-
|
|
135
|
+
solved = solution.task.data
|
|
136
|
+
files = {TRACE_FILE: json.dumps(solution.to_record()).encode()}
|
|
137
|
+
template = config.build_prompt()
|
|
138
|
+
body = _render(template, prompt=solved.prompt_text)
|
|
139
|
+
if "{prompt}" not in template:
|
|
140
|
+
# A policy that doesn't place the task statement itself still needs it.
|
|
141
|
+
body += "\n\n" + _render(TASK_SECTION, prompt=solved.prompt_text)
|
|
142
|
+
sections = [body, _verdict_section(config.criteria()), SANDBOX_NOTE]
|
|
143
|
+
if (hint := config.build_hint()) is not None:
|
|
144
|
+
sections.insert(1, _render(HINT_SECTION, hint=hint))
|
|
145
|
+
prompt = "\n\n".join(sections)
|
|
79
146
|
return cls(
|
|
80
147
|
vf.TaskData(
|
|
81
|
-
idx=
|
|
82
|
-
prompt=prompt
|
|
83
|
-
image=
|
|
84
|
-
workdir=
|
|
85
|
-
resources=
|
|
148
|
+
idx=solved.idx,
|
|
149
|
+
prompt=prompt,
|
|
150
|
+
image=solved.image,
|
|
151
|
+
workdir=solved.workdir,
|
|
152
|
+
resources=solved.resources,
|
|
86
153
|
),
|
|
87
|
-
files=
|
|
88
|
-
TRANSCRIPT_MD: solution.transcript.encode(),
|
|
89
|
-
TRANSCRIPT_JSON: json.dumps(solution.to_record()).encode(),
|
|
90
|
-
},
|
|
154
|
+
files=files,
|
|
91
155
|
)
|
|
92
156
|
|
|
93
157
|
async def setup(self, trace: vf.Trace, runtime: vf.Runtime) -> None:
|
|
94
|
-
#
|
|
95
|
-
# judge's own
|
|
96
|
-
|
|
97
|
-
|
|
158
|
+
# The solver had this box first: a pre-seeded verdict must never read as
|
|
159
|
+
# the judge's own, and a file (or planted symlink) at an upload path must
|
|
160
|
+
# never survive it — a symlinked TRACE_FILE would redirect the write onto
|
|
161
|
+
# any file the solver chose.
|
|
162
|
+
await runtime.run(["rm", "-f", VERDICT_FILE, *self.files], env={})
|
|
163
|
+
for path, content in self.files.items():
|
|
98
164
|
await runtime.write(path, content)
|
|
99
165
|
|
|
100
166
|
async def finalize(self, trace: vf.Trace, runtime: vf.Runtime) -> None:
|
|
@@ -106,39 +172,125 @@ class JudgeTask(vf.Task):
|
|
|
106
172
|
except Exception as e:
|
|
107
173
|
raise ValueError(
|
|
108
174
|
f"the judge wrote no verdict to {VERDICT_FILE}; its final act must "
|
|
109
|
-
'be writing {"
|
|
175
|
+
'be writing {"verdicts": [{"name", "reason", "verdict"}, ...]} there'
|
|
110
176
|
) from e
|
|
111
177
|
trace.info["verdict"] = json.loads(raw)
|
|
112
178
|
|
|
113
179
|
|
|
180
|
+
class JudgeTaskConfig(vf.BaseConfig):
|
|
181
|
+
"""The judge's minted task: the grading policy and what lands in its box."""
|
|
182
|
+
|
|
183
|
+
prompt: Path | str | None = None
|
|
184
|
+
"""Grading-policy override: inline text, or a policy file (a value ending in
|
|
185
|
+
`.md`/`.txt` is read from disk). Replaces only the policy body — the verdict
|
|
186
|
+
contract and workspace note are always appended, so a custom policy cannot
|
|
187
|
+
break verdict scraping. May reference `{prompt}` (the solver task's prompt);
|
|
188
|
+
if it doesn't, the task statement is appended after the policy."""
|
|
189
|
+
hint: Path | str | None = None
|
|
190
|
+
"""Optional hints injected as their own section (inline text or a `.md`/
|
|
191
|
+
`.txt` file): task-family pointers into the trace or box — e.g. for math,
|
|
192
|
+
where the reference answer lives in the record; for SWE, to diff the repo
|
|
193
|
+
or read `info.patch`."""
|
|
194
|
+
rubric: Path | None = None
|
|
195
|
+
"""Criteria the judge grades against: a `.toml`/`.json` file with a
|
|
196
|
+
`criteria` list — the plugged rubric judge's format, so the same rubric
|
|
197
|
+
files work for both. None grades the single built-in `solved` criterion."""
|
|
198
|
+
|
|
199
|
+
@staticmethod
|
|
200
|
+
def _resolve(value: Path | str) -> str:
|
|
201
|
+
path = Path(value)
|
|
202
|
+
if isinstance(value, Path) or path.suffix in (".md", ".txt"):
|
|
203
|
+
return path.read_text(encoding="utf-8")
|
|
204
|
+
return str(value)
|
|
205
|
+
|
|
206
|
+
def build_prompt(self) -> str:
|
|
207
|
+
if self.prompt is None:
|
|
208
|
+
return GRADE_PROMPT + "\n\n" + TASK_SECTION
|
|
209
|
+
return self._resolve(self.prompt)
|
|
210
|
+
|
|
211
|
+
def build_hint(self) -> str | None:
|
|
212
|
+
return self._resolve(self.hint) if self.hint is not None else None
|
|
213
|
+
|
|
214
|
+
def criteria(self) -> list[Criterion]:
|
|
215
|
+
if self.rubric is None:
|
|
216
|
+
return [SOLVED]
|
|
217
|
+
text = self.rubric.read_text(encoding="utf-8")
|
|
218
|
+
data = (
|
|
219
|
+
tomllib.loads(text)
|
|
220
|
+
if self.rubric.suffix.lower() == ".toml"
|
|
221
|
+
else json.loads(text)
|
|
222
|
+
)
|
|
223
|
+
items = data.get("criteria", []) if isinstance(data, dict) else data
|
|
224
|
+
criteria = [Criterion.model_validate(item) for item in items]
|
|
225
|
+
if not criteria:
|
|
226
|
+
raise ValueError(f"rubric file '{self.rubric}' lists no criteria")
|
|
227
|
+
names = [criterion.name for criterion in criteria]
|
|
228
|
+
if len(set(names)) != len(names):
|
|
229
|
+
raise ValueError(
|
|
230
|
+
f"rubric file '{self.rubric}' has duplicate criterion names"
|
|
231
|
+
)
|
|
232
|
+
if bad := [c.name for c in criteria if not 0 <= c.weight < math.inf]:
|
|
233
|
+
raise ValueError(
|
|
234
|
+
f"rubric '{self.rubric}' has negative or non-finite criterion "
|
|
235
|
+
f"weights: {bad}"
|
|
236
|
+
)
|
|
237
|
+
if sum(criterion.weight for criterion in criteria) <= 0:
|
|
238
|
+
raise ValueError(f"rubric '{self.rubric}' has no positive criterion weight")
|
|
239
|
+
return criteria
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
class ScoreConfig(vf.BaseConfig):
|
|
243
|
+
"""How the judge's verdict composes with the taskset's own rewards on the
|
|
244
|
+
solver's trace. Judge-only by default."""
|
|
245
|
+
|
|
246
|
+
task_weight: float = 0.0
|
|
247
|
+
"""Scale applied to the taskset's own rewards; 1 keeps them next to the verdict."""
|
|
248
|
+
judge_weight: float = 1.0
|
|
249
|
+
"""Weight of the judge's verdict in the solver's reward."""
|
|
250
|
+
|
|
251
|
+
|
|
114
252
|
class AgenticJudgeEnvConfig(vf.EnvConfig):
|
|
115
253
|
solver: vf.AgentConfig = vf.AgentConfig()
|
|
254
|
+
"""The solver agent. It owns the shared box, so its runtime must be a
|
|
255
|
+
container: `--env.solver.runtime.type docker|prime`."""
|
|
116
256
|
judge: vf.AgentConfig = vf.AgentConfig()
|
|
117
|
-
"""The judge agent.
|
|
118
|
-
|
|
257
|
+
"""The judge agent. It plays in the solver's box; its own runtime policy is
|
|
258
|
+
ignored (overwritten with the solver's)."""
|
|
259
|
+
task: JudgeTaskConfig = JudgeTaskConfig()
|
|
260
|
+
score: ScoreConfig = ScoreConfig()
|
|
119
261
|
|
|
120
262
|
|
|
121
263
|
class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
|
|
122
264
|
def __init__(self, config: AgenticJudgeEnvConfig) -> None:
|
|
265
|
+
# The judge plays in the solver's box, so its effective runtime IS the
|
|
266
|
+
# solver's policy — aligning the config keeps the base env's subprocess
|
|
267
|
+
# warning and the runtime stamped on the judge's trace truthful.
|
|
268
|
+
config.judge = config.judge.model_copy(
|
|
269
|
+
update={"runtime": config.solver.runtime}
|
|
270
|
+
)
|
|
123
271
|
super().__init__(config)
|
|
124
|
-
self.
|
|
272
|
+
self._check_agents()
|
|
273
|
+
# A missing policy file or a malformed rubric fails here, not mid-episode.
|
|
274
|
+
config.task.build_prompt()
|
|
275
|
+
config.task.build_hint()
|
|
276
|
+
config.task.criteria()
|
|
125
277
|
|
|
126
|
-
def
|
|
278
|
+
def _check_agents(self) -> None:
|
|
127
279
|
"""The judge executes real code, never on the host — refuse an impossible
|
|
128
|
-
|
|
129
|
-
|
|
280
|
+
pairing at construction, not after burning a full solver run."""
|
|
281
|
+
judge = self._harnesses["judge"]
|
|
282
|
+
if not judge.EXECUTES_CODE:
|
|
130
283
|
raise ValueError(
|
|
131
284
|
"agentic-judge plays a code-executing judge in its own sandbox, but "
|
|
132
|
-
f"harness {
|
|
285
|
+
f"harness {judge.config.id!r} is a tool-less chat loop — a verdict "
|
|
133
286
|
"that needs no execution is a plugged judge "
|
|
134
287
|
"(--env.taskset.task.judges), not an agent."
|
|
135
288
|
)
|
|
136
|
-
if isinstance(
|
|
289
|
+
if isinstance(self.config.solver.runtime, vf.SubprocessConfig):
|
|
137
290
|
raise ValueError(
|
|
138
|
-
"agentic-judge plays its judge in
|
|
139
|
-
"
|
|
140
|
-
"
|
|
141
|
-
"or prime."
|
|
291
|
+
"agentic-judge plays its judge in the solver's box, but the solver "
|
|
292
|
+
"(which provisions it) resolves to the subprocess runtime; use "
|
|
293
|
+
"--env.solver.runtime.type docker or prime"
|
|
142
294
|
)
|
|
143
295
|
|
|
144
296
|
async def setup(self, agents: vf.Agents) -> None:
|
|
@@ -146,29 +298,50 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
|
|
|
146
298
|
agents.judge.trainable = False
|
|
147
299
|
|
|
148
300
|
async def run(self, task: vf.Task, agents: vf.Agents) -> None:
|
|
149
|
-
|
|
150
|
-
|
|
301
|
+
async with agents.solver.provision(task) as box:
|
|
302
|
+
solution = await agents.solver.run(task, runtime=box)
|
|
303
|
+
judge_task = JudgeTask.from_trace(solution, self.config.task)
|
|
304
|
+
await agents.judge.run(judge_task, runtime=box)
|
|
151
305
|
|
|
152
306
|
async def finalize(self, task: vf.Task, episode: vf.Episode) -> None:
|
|
153
|
-
"""Record the scraped verdict on the SOLVER's trace. Strict on scale: an
|
|
154
|
-
off-scale score raises (a judge answering `95` must not clamp to full
|
|
155
|
-
marks), failing the episode rather than scoring the solver wrong."""
|
|
156
307
|
by_agent = {t.agent_name: t for t in episode.traces}
|
|
157
308
|
solution, verdict = by_agent["solver"], by_agent["judge"]
|
|
158
309
|
data = verdict.info.get("verdict")
|
|
159
|
-
if not isinstance(data, dict):
|
|
310
|
+
if not isinstance(data, dict) or not isinstance(data.get("verdicts"), list):
|
|
160
311
|
raise ValueError(
|
|
161
|
-
f"no
|
|
312
|
+
f"no verdicts on the judge's trace (expected {VERDICT_FILE} with a "
|
|
313
|
+
'"verdicts" list)'
|
|
162
314
|
)
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
315
|
+
criteria = self.config.task.criteria()
|
|
316
|
+
by_criterion = {c.name: c for c in criteria}
|
|
317
|
+
answers: dict[str, str] = {}
|
|
318
|
+
for entry in data["verdicts"]:
|
|
319
|
+
if not isinstance(entry, dict):
|
|
320
|
+
raise ValueError(f"verdict entry {entry!r} is not an object")
|
|
321
|
+
name = str(entry.get("name"))
|
|
322
|
+
if name in answers:
|
|
323
|
+
# Contradictory duplicates must not collapse to whichever came last.
|
|
324
|
+
raise ValueError(f"judge answered criterion {name!r} more than once")
|
|
325
|
+
answers[name] = str(entry.get("verdict"))
|
|
326
|
+
if sorted(answers) != sorted(by_criterion):
|
|
170
327
|
raise ValueError(
|
|
171
|
-
f"
|
|
172
|
-
"
|
|
328
|
+
f"judge verdicts name {sorted(answers)}, expected the rubric's "
|
|
329
|
+
f"{sorted(by_criterion)}"
|
|
173
330
|
)
|
|
174
|
-
|
|
331
|
+
scores: dict[str, float] = {}
|
|
332
|
+
for name, answer in answers.items():
|
|
333
|
+
choices = by_criterion[name].choices
|
|
334
|
+
# An off-menu answer is a judge failure, not a zero score.
|
|
335
|
+
if answer not in choices:
|
|
336
|
+
raise ValueError(
|
|
337
|
+
f"judge answered {answer!r} for '{name}', expected one of {choices}"
|
|
338
|
+
)
|
|
339
|
+
scores[name] = choices.index(answer) / (len(choices) - 1)
|
|
340
|
+
for criterion in criteria:
|
|
341
|
+
solution.record_metric(f"judge/{criterion.name}", scores[criterion.name])
|
|
342
|
+
if self.config.score.task_weight != 1.0:
|
|
343
|
+
for reward in solution.rewards.values():
|
|
344
|
+
reward.weight *= self.config.score.task_weight
|
|
345
|
+
total = sum(criterion.weight for criterion in criteria)
|
|
346
|
+
reward = sum(c.weight * scores[c.name] for c in criteria) / total
|
|
347
|
+
solution.record_reward("judge", reward, weight=self.config.score.judge_weight)
|
verifiers/v1/legacy.py
CHANGED
|
@@ -37,6 +37,7 @@ from verifiers.v1.trace import (
|
|
|
37
37
|
Error,
|
|
38
38
|
GenerationSpan,
|
|
39
39
|
ModelCall,
|
|
40
|
+
Reward,
|
|
40
41
|
TimeSpan,
|
|
41
42
|
TimeSplit,
|
|
42
43
|
Timing,
|
|
@@ -270,7 +271,7 @@ def rollout_output_to_trace(out: dict, task_idx: int) -> Trace:
|
|
|
270
271
|
data=_to_wire_task(task_idx, out.get("prompt"), out.get("answer")),
|
|
271
272
|
),
|
|
272
273
|
tools=_to_v1_tools(out.get("tool_defs")),
|
|
273
|
-
rewards={"reward": float(out.get("reward") or 0.0)},
|
|
274
|
+
rewards={"reward": Reward(score=float(out.get("reward") or 0.0))},
|
|
274
275
|
metrics={k: float(v) for k, v in (out.get("metrics") or {}).items()},
|
|
275
276
|
info=dict(out.get("info") or {}),
|
|
276
277
|
is_completed=bool(out.get("is_completed", True)),
|
verifiers/v1/push.py
CHANGED
|
@@ -90,9 +90,10 @@ def trace_to_sample(
|
|
|
90
90
|
else None,
|
|
91
91
|
"info": dict(trace.info) or None,
|
|
92
92
|
}
|
|
93
|
-
# Flatten sub-rewards to top-level keys the way v0 does
|
|
94
|
-
|
|
95
|
-
|
|
93
|
+
# Flatten sub-rewards to top-level keys the way v0 does (raw scores, as v0's
|
|
94
|
+
# per-function outputs were); env metrics stay nested.
|
|
95
|
+
for name, reward in trace.rewards.items():
|
|
96
|
+
sample.setdefault(name, reward.score)
|
|
96
97
|
return sample
|
|
97
98
|
|
|
98
99
|
|
|
@@ -127,7 +128,8 @@ def _run_metrics(episodes: list[Episode], traces: list[Trace]) -> dict[str, Any]
|
|
|
127
128
|
sums: dict[str, float] = {}
|
|
128
129
|
counts: dict[str, int] = {}
|
|
129
130
|
for trace in scored:
|
|
130
|
-
for name,
|
|
131
|
+
scores = {name: reward.score for name, reward in trace.rewards.items()}
|
|
132
|
+
for name, value in {**scores, **trace.metrics}.items():
|
|
131
133
|
sums[name] = sums.get(name, 0.0) + value
|
|
132
134
|
counts[name] = counts.get(name, 0) + 1
|
|
133
135
|
n = len(scored)
|
verifiers/v1/trace.py
CHANGED
|
@@ -266,7 +266,7 @@ _NODE_DUMP_EXCLUDE: dict = {
|
|
|
266
266
|
"""Raw tensor fields kept on the msgpack wire but excluded from JSON records."""
|
|
267
267
|
|
|
268
268
|
|
|
269
|
-
TRACE_VERSION =
|
|
269
|
+
TRACE_VERSION = 4
|
|
270
270
|
"""Version of the trace record schema (see `Trace.model_json_schema()`). Bumped on
|
|
271
271
|
breaking shape changes; optional-with-default fields are additive and don't bump it."""
|
|
272
272
|
|
|
@@ -343,6 +343,21 @@ class TraceTask(StrictBaseModel, Generic[DataT]):
|
|
|
343
343
|
"""The (immutable) row being solved."""
|
|
344
344
|
|
|
345
345
|
|
|
346
|
+
class Reward(StrictBaseModel):
|
|
347
|
+
"""One named reward as recorded on the trace: the raw score next to its weight,
|
|
348
|
+
so records keep both readable and the weighted sum stays a derived view."""
|
|
349
|
+
|
|
350
|
+
score: float
|
|
351
|
+
"""The raw value the reward function returned, unweighted."""
|
|
352
|
+
weight: float = 1.0
|
|
353
|
+
"""The multiplier `score` carries in the trace-level `reward` sum."""
|
|
354
|
+
|
|
355
|
+
@property
|
|
356
|
+
def value(self) -> float:
|
|
357
|
+
"""This reward's weighted contribution to the trace-level `reward`."""
|
|
358
|
+
return self.score * self.weight
|
|
359
|
+
|
|
360
|
+
|
|
346
361
|
class Trace(StrictBaseModel, Generic[DataT, StateT, AgentConfigT]):
|
|
347
362
|
id: str = Field(default_factory=lambda: uuid.uuid4().hex)
|
|
348
363
|
"""Unique id for this rollout, auto-generated per trace."""
|
|
@@ -369,8 +384,9 @@ class Trace(StrictBaseModel, Generic[DataT, StateT, AgentConfigT]):
|
|
|
369
384
|
"""Every provider exchange behind the sampled turns, in order: raw wire request/response
|
|
370
385
|
plus per-call timing and errors, linked into `nodes` via `ModelCall.node`."""
|
|
371
386
|
|
|
372
|
-
rewards: dict[str,
|
|
373
|
-
"""
|
|
387
|
+
rewards: dict[str, Reward] = Field(default_factory=dict)
|
|
388
|
+
"""Named rewards from tasks, judges, and the env's `score()` — each keeps its
|
|
389
|
+
raw `score` and `weight`; the trace-level `reward` is their weighted sum."""
|
|
374
390
|
metrics: dict[str, float] = Field(default_factory=dict)
|
|
375
391
|
"""Unweighted metrics from tasks, harnesses, and judges."""
|
|
376
392
|
info: dict[str, Any] = Field(default_factory=dict)
|
|
@@ -400,7 +416,7 @@ class Trace(StrictBaseModel, Generic[DataT, StateT, AgentConfigT]):
|
|
|
400
416
|
|
|
401
417
|
@property
|
|
402
418
|
def reward(self) -> float:
|
|
403
|
-
return sum(self.rewards.values())
|
|
419
|
+
return sum(r.value for r in self.rewards.values())
|
|
404
420
|
|
|
405
421
|
@property
|
|
406
422
|
def error(self) -> Error | None:
|
|
@@ -559,12 +575,12 @@ class Trace(StrictBaseModel, Generic[DataT, StateT, AgentConfigT]):
|
|
|
559
575
|
self.extra_usage.append(response.usage)
|
|
560
576
|
|
|
561
577
|
def record_reward(self, name: str, value: float, weight: float = 1.0) -> None:
|
|
562
|
-
|
|
578
|
+
reward = Reward(score=float(value), weight=float(weight))
|
|
563
579
|
if name in self.rewards:
|
|
564
580
|
logger.warning(
|
|
565
|
-
"reward %r overridden: %s -> %s", name, self.rewards[name],
|
|
581
|
+
"reward %r overridden: %s -> %s", name, self.rewards[name], reward
|
|
566
582
|
)
|
|
567
|
-
self.rewards[name] =
|
|
583
|
+
self.rewards[name] = reward
|
|
568
584
|
|
|
569
585
|
def stamp(self, run: RunInfo | None = None, **info: Any) -> None:
|
|
570
586
|
"""Stamp identity only the consumer knows (the eval CLI / a trainer) onto the
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.dev27
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -167,7 +167,7 @@ verifiers/utils/threaded_sandbox_client.py,sha256=Pbr8MA4FDPEitL6z88S8T1qLJOmtXN
|
|
|
167
167
|
verifiers/utils/tool_utils.py,sha256=gInWZODWQUZN4TyEMuuITyOUm9qlHJxtcvr1zZl_zJs,1020
|
|
168
168
|
verifiers/utils/usage_utils.py,sha256=GPLC0xGY_Obrr8X7huWY-2iODZ7tgX-MtO8WSDN4rXI,3904
|
|
169
169
|
verifiers/utils/version_utils.py,sha256=am3hZLnlaUWTFdllGZR1VMN91hEhiiE8FDmPf1cOG1k,2642
|
|
170
|
-
verifiers/v1/__init__.py,sha256=
|
|
170
|
+
verifiers/v1/__init__.py,sha256=z0yUgI1iVf8znsAtBG2fpRKuwipaQ5yS4PvGhWJbKtE,6859
|
|
171
171
|
verifiers/v1/agent.py,sha256=FP38GsZXuEJe3tJ-U2THpwBf9NWQeo847hWRZoeggNw,31956
|
|
172
172
|
verifiers/v1/decorators.py,sha256=XRMkUQSyvXCYP5fOwzBYV5qEOlxLg6rzYhnhrHHHTIQ,3838
|
|
173
173
|
verifiers/v1/env.py,sha256=w6sHWdWijLRZzhwVyJjadnGtziNqJAPCeMUoNdzHoLg,17712
|
|
@@ -176,9 +176,9 @@ verifiers/v1/errors.py,sha256=Pj5Om8x1fP3TDPQQn6nUCoSPoZRsy2JeBz8pXhpPrDY,6883
|
|
|
176
176
|
verifiers/v1/graph.py,sha256=Mivq5jICUcyK-fhomFTwP4fQn688MXg6-hmdf03aC4Y,27703
|
|
177
177
|
verifiers/v1/harness.py,sha256=Ukzhk7wxSUSVnN4tRAUMfniN9R9MBSNzs2SiT5YcRzQ,10711
|
|
178
178
|
verifiers/v1/judge.py,sha256=ZWvr6uCniSwqzzjKrlyh-lzfCh5q-VyB16W42x0aWlg,9449
|
|
179
|
-
verifiers/v1/legacy.py,sha256=
|
|
179
|
+
verifiers/v1/legacy.py,sha256=8eVGhutQEgJG4qabhph4Xs1VzTWvmMxMn3-Zk-6ViWM,22306
|
|
180
180
|
verifiers/v1/loaders.py,sha256=FdICekH_c9WYkUe0XJCI9AWxL1sJjjzzJPJ7ZJdeyRk,9962
|
|
181
|
-
verifiers/v1/push.py,sha256=
|
|
181
|
+
verifiers/v1/push.py,sha256=VfLESKAlL8WXzfXx6CrqwMJ-CvPTkP-tUf6Rin12V8w,11273
|
|
182
182
|
verifiers/v1/retries.py,sha256=ZQxY6R_FoXooERmhIMZnhY3b2koJOVjaIvkSO9gw53E,5379
|
|
183
183
|
verifiers/v1/rollout.py,sha256=8_LC938Ws9uWFRPz2DDpivttqDALb3Yir7J5wjnsoIE,20332
|
|
184
184
|
verifiers/v1/scoring.py,sha256=I_mtqhTZ195hj2-CB5jzr3BY5GFKgNEvTf331f7Is4k,5740
|
|
@@ -186,7 +186,7 @@ verifiers/v1/session.py,sha256=J6ZPuvsg6vMTJAs9dMUMtfdjCuvp6Hom877DVdKFKlY,6912
|
|
|
186
186
|
verifiers/v1/state.py,sha256=R8tyQv8nsFV2pztrquDqCOa1Mk1fAp3w2GjqUugZwE0,689
|
|
187
187
|
verifiers/v1/task.py,sha256=qKAAF154ykgvrms2Y2Or6kgGVPvBMsFCD6KqX2QOQaU,10912
|
|
188
188
|
verifiers/v1/taskset.py,sha256=PX1-skAVhSpGF07qjxcG0Ctmq1LiMftMxxok-uc8qRY,3939
|
|
189
|
-
verifiers/v1/trace.py,sha256=
|
|
189
|
+
verifiers/v1/trace.py,sha256=OE7vW6sYGjA18JdRJuHM9UiRaq9XkGS0Yv2m7nuvEYs,26619
|
|
190
190
|
verifiers/v1/types.py,sha256=5tZyG4r4bLJ9a14oybHy7bSCpt5X5A17eRwmIszuRo8,9118
|
|
191
191
|
verifiers/v1/acp/__init__.py,sha256=9dwH6fLopNndRmcpRSYC8ghzzMquaLfgh645QM-Dwic,2189
|
|
192
192
|
verifiers/v1/acp/_runner.py,sha256=FqEnHR0Q_YrCt_EuaJqiqTpd2_kFa41FqcEunwOhvTI,6856
|
|
@@ -201,7 +201,7 @@ verifiers/v1/cli/serve.py,sha256=VHzcr2bM8R3XoGEu6WvY1NSQoiBZsk49cVW5AB9N4Kg,266
|
|
|
201
201
|
verifiers/v1/cli/validate.py,sha256=r3ByzcLc_DFkvlCB7NanYRnBSOIi5EU9cUANy74l3T0,9547
|
|
202
202
|
verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
|
|
203
203
|
verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
|
|
204
|
-
verifiers/v1/cli/dashboard/eval.py,sha256=
|
|
204
|
+
verifiers/v1/cli/dashboard/eval.py,sha256=nmaWIq-4ormHZWjwlWm1QSFo383Sc4VFEkw9rYnpi00,34357
|
|
205
205
|
verifiers/v1/cli/dashboard/replay.py,sha256=CvRVaf0dUum5v0IzPQbFPPikBQmTrkMeaunGfVMTWHc,2755
|
|
206
206
|
verifiers/v1/cli/dashboard/validate.py,sha256=rLsQ_31DjTIZpC_VO7zSG41ncUEBxqCx8mL9nTnGiMc,3650
|
|
207
207
|
verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
@@ -235,8 +235,8 @@ verifiers/v1/dialects/base.py,sha256=YZQDnasQseDXK-GHLfJLZgyHIIygS80YhD_sTjmUbfA
|
|
|
235
235
|
verifiers/v1/dialects/chat.py,sha256=DFvjmIW86jZUMWuz-hOqzxC6tiiFibiAzNEJ3JkpvjU,13998
|
|
236
236
|
verifiers/v1/dialects/responses.py,sha256=vVgeT7-aWjtfED2IKlDwTATKn_AQZruHNL6mkFTTZTU,13548
|
|
237
237
|
verifiers/v1/envs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
238
|
-
verifiers/v1/envs/agentic_judge/__init__.py,sha256=
|
|
239
|
-
verifiers/v1/envs/agentic_judge/env.py,sha256=
|
|
238
|
+
verifiers/v1/envs/agentic_judge/__init__.py,sha256=X7vQbbbfmJ_j-mJEfBkuB4a8QUPkUYkpG7yp0xYOEmg,279
|
|
239
|
+
verifiers/v1/envs/agentic_judge/env.py,sha256=_zEKG9i3LsOVIQWULXYAHpbmrQNPWBRayJx2yjvtPbQ,15307
|
|
240
240
|
verifiers/v1/envs/best_of_n/__init__.py,sha256=mubASiwXMNIhnQFDccNAc7yZ0OFsOVK8hH8WIx9b2A4,119
|
|
241
241
|
verifiers/v1/envs/best_of_n/env.py,sha256=KEUJBCVqblnXP_lEwm3mjGdaikBMkecz8UjOFPBhxc0,1987
|
|
242
242
|
verifiers/v1/envs/single_agent/__init__.py,sha256=r0edAc_6gtHBhPzeBGVcdHZZqSdCPZnLMje6eVgyz5U,138
|
|
@@ -327,8 +327,8 @@ verifiers/v1/utils/logging.py,sha256=OcMHA6NsYux3oIzjPuI95rDWmFBHNcHDjZeNIhXTX-Y
|
|
|
327
327
|
verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
|
|
328
328
|
verifiers/v1/utils/sampling.py,sha256=JczGzBn6s3wsIrSY2Hy3hE8m6NreNgsNzhSjDikQqX0,1037
|
|
329
329
|
verifiers/v1/utils/version.py,sha256=75ZtI2NHBmlb52KpcKLUpiSXp8q4bASr7uKeXoCKlT8,1582
|
|
330
|
-
verifiers-0.2.2.
|
|
331
|
-
verifiers-0.2.2.
|
|
332
|
-
verifiers-0.2.2.
|
|
333
|
-
verifiers-0.2.2.
|
|
334
|
-
verifiers-0.2.2.
|
|
330
|
+
verifiers-0.2.2.dev27.dist-info/METADATA,sha256=JnRcvyGkm0iogzBO_JnbUyqr6DO1_EUfDXi-TCV6ezg,4540
|
|
331
|
+
verifiers-0.2.2.dev27.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
332
|
+
verifiers-0.2.2.dev27.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
|
|
333
|
+
verifiers-0.2.2.dev27.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
334
|
+
verifiers-0.2.2.dev27.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|