verifiers 0.2.2.dev25__py3-none-any.whl → 0.2.2.dev27__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
verifiers/v1/__init__.py CHANGED
@@ -107,6 +107,7 @@ from verifiers.v1.trace import (
107
107
  EvalRunInfo,
108
108
  GenerationSpan,
109
109
  ModelCall,
110
+ Reward,
110
111
  RunInfo,
111
112
  TimeSpan,
112
113
  TimeSplit,
@@ -171,6 +172,7 @@ __all__ = [
171
172
  "Trace",
172
173
  "TraceTask",
173
174
  "WireTrace",
175
+ "Reward",
174
176
  "Episode",
175
177
  "WireEpisode",
176
178
  "TRACE_VERSION",
@@ -341,13 +341,19 @@ def _score_segments(traces: list[Trace], source: str) -> str | None:
341
341
  return None
342
342
  segments = []
343
343
  for name in names:
344
- mean = format_mean(
345
- traces, lambda t, n=name, s=source: getattr(t, s).get(n, 0.0)
346
- )
344
+ mean = format_mean(traces, lambda t, n=name, s=source: _score(t, s, n))
347
345
  segments.append(f"{name} {mean}")
348
346
  return " · ".join(segments)
349
347
 
350
348
 
349
+ def _score(trace: Trace, source: str, name: str) -> float:
350
+ """Rewards carry raw score + weight; the breakdown shows the raw score."""
351
+ if source == "rewards":
352
+ reward = trace.rewards.get(name)
353
+ return reward.score if reward is not None else 0.0
354
+ return trace.metrics.get(name, 0.0)
355
+
356
+
351
357
  def _breakdown(scored: list[Trace], done: list[Trace]) -> Table | None:
352
358
  """Score rows read the policy view (`scored` — trainable traces); with several
353
359
  roles in play they split per role, each role averaging over its OWN traces (no
@@ -1,3 +1,15 @@
1
- from verifiers.v1.envs.agentic_judge.env import AgenticJudgeEnv, AgenticJudgeEnvConfig
1
+ from verifiers.v1.envs.agentic_judge.env import (
2
+ AgenticJudgeEnv,
3
+ AgenticJudgeEnvConfig,
4
+ Criterion,
5
+ JudgeTaskConfig,
6
+ ScoreConfig,
7
+ )
2
8
 
3
- __all__ = ["AgenticJudgeEnv", "AgenticJudgeEnvConfig"]
9
+ __all__ = [
10
+ "AgenticJudgeEnv",
11
+ "AgenticJudgeEnvConfig",
12
+ "Criterion",
13
+ "JudgeTaskConfig",
14
+ "ScoreConfig",
15
+ ]
@@ -1,100 +1,166 @@
1
- """agentic-judge: a solver plays the task, a code-executing judge verifies it in a sandbox.
2
-
3
- Agent-as-judge as a reusable env (`--env.id agentic-judge` over any taskset). The
4
- judge's task mirrors the solver task's world — same image/workdir/resources, in a
5
- FRESH box in its original state (the solver's runtime is gone by judge time) —
6
- with the graded transcript uploaded, so the judge reconstructs and tests the work
7
- empirically, always in its own sandbox, never on the host.
8
-
9
- The verdict channel is a file, not the chat: the judge writes
10
- `{"score": 0-10, "reasoning": ...}` to `/tmp/verdict.json` in its box (a file
11
- survives a chatty final reply), `JudgeTask.finalize` scrapes it off the live
12
- runtime onto the judge's trace, and the env's `finalize()` validates it strictly
13
- onto the solver's trace — a missing, malformed, or off-scale verdict fails loudly instead
14
- of clamping to full marks.
1
+ """agentic-judge: a solver plays the task, a judge verifies it in the same box.
2
+
3
+ A reusable env (`--env.id agentic-judge` over any taskset): the box is
4
+ provisioned from the solver's runtime policy, the solver plays the task in it,
5
+ and a code-executing judge then inspects the work as the agent left it, with
6
+ the solver's full trace record uploaded at `/tmp/trace.json`. The judge grades
7
+ rubric criteria (`[env.task]`: policy prompt, criteria file) and writes its
8
+ verdicts to `/tmp/verdict.json`; `finalize()` validates them strictly onto the
9
+ solver's trace — `judge/<name>` metrics plus a weighted-mean `judge` reward,
10
+ composed with the taskset's own rewards via `[env.score]` (judge-only by
11
+ default).
15
12
  """
16
13
 
17
14
  import json
18
15
  import math
16
+ import re
17
+ import tomllib
18
+ from pathlib import Path
19
+
20
+ from pydantic import field_validator
19
21
 
20
22
  import verifiers.v1 as vf
21
- from verifiers.v1.harness import Harness
23
+ from verifiers.v1.types import StrictBaseModel
22
24
 
23
- TRANSCRIPT_MD = "/tmp/transcript.md"
24
- TRANSCRIPT_JSON = "/tmp/transcript.json"
25
25
  VERDICT_FILE = "/tmp/verdict.json"
26
+ TRACE_FILE = "/tmp/trace.json"
26
27
 
27
- GRADE_PROMPT = f"""\
28
+ GRADE_PROMPT = """\
28
29
  You are grading another agent's attempt at a task. Verify the work EMPIRICALLY:
29
- reconstruct what the agent did from its transcript and test it with real
30
- execution in your sandbox — never take the transcript's word for an outcome you
31
- can check.
30
+ reconstruct what the agent did from its trace and test it with real execution
31
+ in your sandbox — never take the trace's word for an outcome you can check."""
32
32
 
33
+ TASK_SECTION = """\
33
34
  ## The task the agent was given
34
35
 
35
- {{prompt}}
36
+ {prompt}"""
37
+
38
+
39
+ class Criterion(StrictBaseModel):
40
+ """One rubric criterion — the plugged rubric judge's format, mirrored so the
41
+ same `criteria` files grade both judges."""
42
+
43
+ name: str
44
+ """Key for the criterion's metric (`judge/<name>`)."""
45
+ text: str
46
+ weight: float = 1.0
47
+ """The criterion's share of the reward."""
48
+ choices: list[str] = ["no", "yes"]
49
+ """Allowed answers, ordered **worst → best**: the first scores 0.0, the last 1.0, the rest
50
+ evenly spaced by rank. Default `["no", "yes"]` is a binary check. Needs >= 2, no duplicates."""
51
+
52
+ @field_validator("choices")
53
+ @classmethod
54
+ def _check_choices(cls, v: list[str]) -> list[str]:
55
+ if len(v) < 2:
56
+ raise ValueError(f"`choices` needs at least two options, got {v}")
57
+ if len(set(v)) != len(v):
58
+ raise ValueError(f"`choices` has duplicate options: {v}")
59
+ return v
60
+
61
+
62
+ SOLVED = Criterion(
63
+ name="solved",
64
+ text="The task is fully solved: what the task asked for is achieved, and "
65
+ "you verified it with real execution.",
66
+ )
36
67
 
68
+
69
+ def _verdict_section(criteria: list[Criterion]) -> str:
70
+ listing = "\n".join(
71
+ f"- {c.name}: {c.text} (answer one of, worst to best: {', '.join(c.choices)})"
72
+ for c in criteria
73
+ )
74
+ return f"""\
37
75
  ## Your verdict
38
76
 
77
+ Grade the attempt on these criteria:
78
+
79
+ {listing}
80
+
39
81
  When you are done verifying, write your verdict as JSON to `{VERDICT_FILE}`:
40
82
 
41
- {{{{"score": <integer 0-10>, "reasoning": "<one paragraph>"}}}}
83
+ {{"verdicts": [{{"name": "<criterion name>", "reason": "<one sentence citing \
84
+ what you verified>", "verdict": "<answer>"}}, ...]}}
42
85
 
43
- 10 = the task is fully solved (you verified it); 0 = no progress. The score MUST
44
- be an integer between 0 and 10 — nothing else is accepted."""
86
+ with one entry per criterion, using each criterion's exact name. For each, first
87
+ write the one-sentence reason grounded in what you actually verified, then set
88
+ verdict to exactly one of the options listed in parentheses after that
89
+ criterion."""
45
90
 
46
91
 
47
- def _sandbox_note(solver: vf.TaskData) -> str:
48
- """What an agentic judge must know about its box before it starts verifying."""
49
- world = (
50
- f"a fresh instance of the same environment the graded agent worked in "
51
- f"(image {solver.image}), in its ORIGINAL state — the agent's edits are "
52
- "NOT applied; reconstruct them from the transcript to verify"
53
- if solver.image is not None
54
- else "your own — the graded agent worked elsewhere"
55
- )
56
- return (
57
- f"\n\n## Your workspace\nYour sandbox is {world}. The agent's full transcript "
58
- f"is uploaded at {TRANSCRIPT_MD} (rendered) and {TRANSCRIPT_JSON} (the raw "
59
- "trace record)."
60
- )
92
+ def _render(template: str, **fields: str) -> str:
93
+ """Substitute documented placeholders in one pass over the original template —
94
+ str.format would crash on any literal brace in a custom prompt, and sequential
95
+ replaces would re-scan substituted values. An unknown placeholder stays as written."""
96
+ pattern = re.compile(r"\{(" + "|".join(map(re.escape, fields)) + r")\}")
97
+ return pattern.sub(lambda m: fields[m.group(1)], template)
98
+
99
+
100
+ SANDBOX_NOTE = f"""\
101
+ ## Your workspace
102
+
103
+ Your sandbox is the SAME box the graded agent worked in, in the state the agent
104
+ left it — its edits (and any scoring side effects) are applied. The agent's raw
105
+ trace record (JSON: messages, tool calls, and its `info` artifacts) is uploaded
106
+ at `{TRACE_FILE}`. The record can be very large — never dump it whole; peek
107
+ selectively (list its keys, then slice out specific fields with python or jq)
108
+ and pull only what you need. It is complete — it may also carry the task's own
109
+ scores/metrics and reference material (a gold answer, a reference solution,
110
+ held-out tests). Those are context, not your standard: recorded scores can be
111
+ wrong and references can be narrower than the task; do not over-index on how a
112
+ reference solves it. Your verdict is what YOU verified by execution."""
113
+
114
+ HINT_SECTION = """\
115
+ ## Hints
116
+
117
+ {hint}"""
61
118
 
62
119
 
63
120
  class JudgeTask(vf.Task):
64
121
  """The judge's verdict task: the solver task's world mirrored onto the minted
65
- row, transcript uploaded (and any stale verdict removed) before the judge
66
- starts, verdict scraped off the live box after it exits. `NEEDS_CONTAINER`
122
+ row, the trace record written (and any stale verdict removed) before the
123
+ judge starts, verdict scraped off the live box after it exits. `NEEDS_CONTAINER`
67
124
  keeps `Agent.run`'s per-task backstop aligned with the judge's declared need."""
68
125
 
69
126
  NEEDS_CONTAINER = True
70
127
 
71
128
  def __init__(self, data: vf.TaskData, files: dict[str, bytes]) -> None:
72
129
  super().__init__(data)
73
- self._files = files
130
+ self.files = files
74
131
 
75
132
  @classmethod
76
- def from_trace(cls, task: vf.Task, solution: vf.Trace) -> "JudgeTask":
133
+ def from_trace(cls, solution: vf.Trace, config: "JudgeTaskConfig") -> "JudgeTask":
77
134
  """Mint the judge's task from the solver's finished trace."""
78
- prompt = GRADE_PROMPT.format(prompt=task.data.prompt_text)
135
+ solved = solution.task.data
136
+ files = {TRACE_FILE: json.dumps(solution.to_record()).encode()}
137
+ template = config.build_prompt()
138
+ body = _render(template, prompt=solved.prompt_text)
139
+ if "{prompt}" not in template:
140
+ # A policy that doesn't place the task statement itself still needs it.
141
+ body += "\n\n" + _render(TASK_SECTION, prompt=solved.prompt_text)
142
+ sections = [body, _verdict_section(config.criteria()), SANDBOX_NOTE]
143
+ if (hint := config.build_hint()) is not None:
144
+ sections.insert(1, _render(HINT_SECTION, hint=hint))
145
+ prompt = "\n\n".join(sections)
79
146
  return cls(
80
147
  vf.TaskData(
81
- idx=task.data.idx,
82
- prompt=prompt + _sandbox_note(task.data),
83
- image=task.data.image,
84
- workdir=task.data.workdir,
85
- resources=task.data.resources,
148
+ idx=solved.idx,
149
+ prompt=prompt,
150
+ image=solved.image,
151
+ workdir=solved.workdir,
152
+ resources=solved.resources,
86
153
  ),
87
- files={
88
- TRANSCRIPT_MD: solution.transcript.encode(),
89
- TRANSCRIPT_JSON: json.dumps(solution.to_record()).encode(),
90
- },
154
+ files=files,
91
155
  )
92
156
 
93
157
  async def setup(self, trace: vf.Trace, runtime: vf.Runtime) -> None:
94
- # A pre-seeded verdict (baked into the image) must never read as the
95
- # judge's own; remove it before the judge starts.
96
- await runtime.run(["rm", "-f", VERDICT_FILE], env={})
97
- for path, content in self._files.items():
158
+ # The solver had this box first: a pre-seeded verdict must never read as
159
+ # the judge's own, and a file (or planted symlink) at an upload path must
160
+ # never survive it — a symlinked TRACE_FILE would redirect the write onto
161
+ # any file the solver chose.
162
+ await runtime.run(["rm", "-f", VERDICT_FILE, *self.files], env={})
163
+ for path, content in self.files.items():
98
164
  await runtime.write(path, content)
99
165
 
100
166
  async def finalize(self, trace: vf.Trace, runtime: vf.Runtime) -> None:
@@ -106,39 +172,125 @@ class JudgeTask(vf.Task):
106
172
  except Exception as e:
107
173
  raise ValueError(
108
174
  f"the judge wrote no verdict to {VERDICT_FILE}; its final act must "
109
- 'be writing {"score": <0-10>, "reasoning": ...} there'
175
+ 'be writing {"verdicts": [{"name", "reason", "verdict"}, ...]} there'
110
176
  ) from e
111
177
  trace.info["verdict"] = json.loads(raw)
112
178
 
113
179
 
180
+ class JudgeTaskConfig(vf.BaseConfig):
181
+ """The judge's minted task: the grading policy and what lands in its box."""
182
+
183
+ prompt: Path | str | None = None
184
+ """Grading-policy override: inline text, or a policy file (a value ending in
185
+ `.md`/`.txt` is read from disk). Replaces only the policy body — the verdict
186
+ contract and workspace note are always appended, so a custom policy cannot
187
+ break verdict scraping. May reference `{prompt}` (the solver task's prompt);
188
+ if it doesn't, the task statement is appended after the policy."""
189
+ hint: Path | str | None = None
190
+ """Optional hints injected as their own section (inline text or a `.md`/
191
+ `.txt` file): task-family pointers into the trace or box — e.g. for math,
192
+ where the reference answer lives in the record; for SWE, to diff the repo
193
+ or read `info.patch`."""
194
+ rubric: Path | None = None
195
+ """Criteria the judge grades against: a `.toml`/`.json` file with a
196
+ `criteria` list — the plugged rubric judge's format, so the same rubric
197
+ files work for both. None grades the single built-in `solved` criterion."""
198
+
199
+ @staticmethod
200
+ def _resolve(value: Path | str) -> str:
201
+ path = Path(value)
202
+ if isinstance(value, Path) or path.suffix in (".md", ".txt"):
203
+ return path.read_text(encoding="utf-8")
204
+ return str(value)
205
+
206
+ def build_prompt(self) -> str:
207
+ if self.prompt is None:
208
+ return GRADE_PROMPT + "\n\n" + TASK_SECTION
209
+ return self._resolve(self.prompt)
210
+
211
+ def build_hint(self) -> str | None:
212
+ return self._resolve(self.hint) if self.hint is not None else None
213
+
214
+ def criteria(self) -> list[Criterion]:
215
+ if self.rubric is None:
216
+ return [SOLVED]
217
+ text = self.rubric.read_text(encoding="utf-8")
218
+ data = (
219
+ tomllib.loads(text)
220
+ if self.rubric.suffix.lower() == ".toml"
221
+ else json.loads(text)
222
+ )
223
+ items = data.get("criteria", []) if isinstance(data, dict) else data
224
+ criteria = [Criterion.model_validate(item) for item in items]
225
+ if not criteria:
226
+ raise ValueError(f"rubric file '{self.rubric}' lists no criteria")
227
+ names = [criterion.name for criterion in criteria]
228
+ if len(set(names)) != len(names):
229
+ raise ValueError(
230
+ f"rubric file '{self.rubric}' has duplicate criterion names"
231
+ )
232
+ if bad := [c.name for c in criteria if not 0 <= c.weight < math.inf]:
233
+ raise ValueError(
234
+ f"rubric '{self.rubric}' has negative or non-finite criterion "
235
+ f"weights: {bad}"
236
+ )
237
+ if sum(criterion.weight for criterion in criteria) <= 0:
238
+ raise ValueError(f"rubric '{self.rubric}' has no positive criterion weight")
239
+ return criteria
240
+
241
+
242
+ class ScoreConfig(vf.BaseConfig):
243
+ """How the judge's verdict composes with the taskset's own rewards on the
244
+ solver's trace. Judge-only by default."""
245
+
246
+ task_weight: float = 0.0
247
+ """Scale applied to the taskset's own rewards; 1 keeps them next to the verdict."""
248
+ judge_weight: float = 1.0
249
+ """Weight of the judge's verdict in the solver's reward."""
250
+
251
+
114
252
  class AgenticJudgeEnvConfig(vf.EnvConfig):
115
253
  solver: vf.AgentConfig = vf.AgentConfig()
254
+ """The solver agent. It owns the shared box, so its runtime must be a
255
+ container: `--env.solver.runtime.type docker|prime`."""
116
256
  judge: vf.AgentConfig = vf.AgentConfig()
117
- """The judge agent. Its runtime must be a container:
118
- `--env.judge.runtime.type docker|prime`."""
257
+ """The judge agent. It plays in the solver's box; its own runtime policy is
258
+ ignored (overwritten with the solver's)."""
259
+ task: JudgeTaskConfig = JudgeTaskConfig()
260
+ score: ScoreConfig = ScoreConfig()
119
261
 
120
262
 
121
263
  class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
122
264
  def __init__(self, config: AgenticJudgeEnvConfig) -> None:
265
+ # The judge plays in the solver's box, so its effective runtime IS the
266
+ # solver's policy — aligning the config keeps the base env's subprocess
267
+ # warning and the runtime stamped on the judge's trace truthful.
268
+ config.judge = config.judge.model_copy(
269
+ update={"runtime": config.solver.runtime}
270
+ )
123
271
  super().__init__(config)
124
- self._check_judge(self._harnesses["judge"], config.judge)
272
+ self._check_agents()
273
+ # A missing policy file or a malformed rubric fails here, not mid-episode.
274
+ config.task.build_prompt()
275
+ config.task.build_hint()
276
+ config.task.criteria()
125
277
 
126
- def _check_judge(self, harness: Harness, judge: vf.AgentConfig) -> None:
278
+ def _check_agents(self) -> None:
127
279
  """The judge executes real code, never on the host — refuse an impossible
128
- judge at construction, not after burning a full solver run."""
129
- if not harness.EXECUTES_CODE:
280
+ pairing at construction, not after burning a full solver run."""
281
+ judge = self._harnesses["judge"]
282
+ if not judge.EXECUTES_CODE:
130
283
  raise ValueError(
131
284
  "agentic-judge plays a code-executing judge in its own sandbox, but "
132
- f"harness {harness.config.id!r} is a tool-less chat loop — a verdict "
285
+ f"harness {judge.config.id!r} is a tool-less chat loop — a verdict "
133
286
  "that needs no execution is a plugged judge "
134
287
  "(--env.taskset.task.judges), not an agent."
135
288
  )
136
- if isinstance(judge.runtime, vf.SubprocessConfig):
289
+ if isinstance(self.config.solver.runtime, vf.SubprocessConfig):
137
290
  raise ValueError(
138
- "agentic-judge plays its judge in a container (JudgeTask mirrors "
139
- "the solver task's image), but the judge resolves to the "
140
- "subprocess runtime; use --env.judge.runtime.type docker "
141
- "or prime."
291
+ "agentic-judge plays its judge in the solver's box, but the solver "
292
+ "(which provisions it) resolves to the subprocess runtime; use "
293
+ "--env.solver.runtime.type docker or prime"
142
294
  )
143
295
 
144
296
  async def setup(self, agents: vf.Agents) -> None:
@@ -146,29 +298,50 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
146
298
  agents.judge.trainable = False
147
299
 
148
300
  async def run(self, task: vf.Task, agents: vf.Agents) -> None:
149
- solution = await agents.solver.run(task)
150
- await agents.judge.run(JudgeTask.from_trace(task, solution))
301
+ async with agents.solver.provision(task) as box:
302
+ solution = await agents.solver.run(task, runtime=box)
303
+ judge_task = JudgeTask.from_trace(solution, self.config.task)
304
+ await agents.judge.run(judge_task, runtime=box)
151
305
 
152
306
  async def finalize(self, task: vf.Task, episode: vf.Episode) -> None:
153
- """Record the scraped verdict on the SOLVER's trace. Strict on scale: an
154
- off-scale score raises (a judge answering `95` must not clamp to full
155
- marks), failing the episode rather than scoring the solver wrong."""
156
307
  by_agent = {t.agent_name: t for t in episode.traces}
157
308
  solution, verdict = by_agent["solver"], by_agent["judge"]
158
309
  data = verdict.info.get("verdict")
159
- if not isinstance(data, dict):
310
+ if not isinstance(data, dict) or not isinstance(data.get("verdicts"), list):
160
311
  raise ValueError(
161
- f"no verdict on the judge's trace (expected {VERDICT_FILE})"
312
+ f"no verdicts on the judge's trace (expected {VERDICT_FILE} with a "
313
+ '"verdicts" list)'
162
314
  )
163
- score = data.get("score")
164
- if (
165
- not isinstance(score, (int, float))
166
- or isinstance(score, bool)
167
- or math.isnan(float(score))
168
- or not 0 <= float(score) <= 10
169
- ):
315
+ criteria = self.config.task.criteria()
316
+ by_criterion = {c.name: c for c in criteria}
317
+ answers: dict[str, str] = {}
318
+ for entry in data["verdicts"]:
319
+ if not isinstance(entry, dict):
320
+ raise ValueError(f"verdict entry {entry!r} is not an object")
321
+ name = str(entry.get("name"))
322
+ if name in answers:
323
+ # Contradictory duplicates must not collapse to whichever came last.
324
+ raise ValueError(f"judge answered criterion {name!r} more than once")
325
+ answers[name] = str(entry.get("verdict"))
326
+ if sorted(answers) != sorted(by_criterion):
170
327
  raise ValueError(
171
- f"verdict score {score!r} is not on the 0-10 scale; refusing to "
172
- "clamp or coerce it"
328
+ f"judge verdicts name {sorted(answers)}, expected the rubric's "
329
+ f"{sorted(by_criterion)}"
173
330
  )
174
- solution.record_reward("judge", float(score) / 10.0)
331
+ scores: dict[str, float] = {}
332
+ for name, answer in answers.items():
333
+ choices = by_criterion[name].choices
334
+ # An off-menu answer is a judge failure, not a zero score.
335
+ if answer not in choices:
336
+ raise ValueError(
337
+ f"judge answered {answer!r} for '{name}', expected one of {choices}"
338
+ )
339
+ scores[name] = choices.index(answer) / (len(choices) - 1)
340
+ for criterion in criteria:
341
+ solution.record_metric(f"judge/{criterion.name}", scores[criterion.name])
342
+ if self.config.score.task_weight != 1.0:
343
+ for reward in solution.rewards.values():
344
+ reward.weight *= self.config.score.task_weight
345
+ total = sum(criterion.weight for criterion in criteria)
346
+ reward = sum(c.weight * scores[c.name] for c in criteria) / total
347
+ solution.record_reward("judge", reward, weight=self.config.score.judge_weight)
verifiers/v1/legacy.py CHANGED
@@ -37,6 +37,7 @@ from verifiers.v1.trace import (
37
37
  Error,
38
38
  GenerationSpan,
39
39
  ModelCall,
40
+ Reward,
40
41
  TimeSpan,
41
42
  TimeSplit,
42
43
  Timing,
@@ -270,7 +271,7 @@ def rollout_output_to_trace(out: dict, task_idx: int) -> Trace:
270
271
  data=_to_wire_task(task_idx, out.get("prompt"), out.get("answer")),
271
272
  ),
272
273
  tools=_to_v1_tools(out.get("tool_defs")),
273
- rewards={"reward": float(out.get("reward") or 0.0)},
274
+ rewards={"reward": Reward(score=float(out.get("reward") or 0.0))},
274
275
  metrics={k: float(v) for k, v in (out.get("metrics") or {}).items()},
275
276
  info=dict(out.get("info") or {}),
276
277
  is_completed=bool(out.get("is_completed", True)),
verifiers/v1/push.py CHANGED
@@ -90,9 +90,10 @@ def trace_to_sample(
90
90
  else None,
91
91
  "info": dict(trace.info) or None,
92
92
  }
93
- # Flatten sub-rewards to top-level keys the way v0 does; env metrics stay nested.
94
- for name, value in trace.rewards.items():
95
- sample.setdefault(name, value)
93
+ # Flatten sub-rewards to top-level keys the way v0 does (raw scores, as v0's
94
+ # per-function outputs were); env metrics stay nested.
95
+ for name, reward in trace.rewards.items():
96
+ sample.setdefault(name, reward.score)
96
97
  return sample
97
98
 
98
99
 
@@ -127,7 +128,8 @@ def _run_metrics(episodes: list[Episode], traces: list[Trace]) -> dict[str, Any]
127
128
  sums: dict[str, float] = {}
128
129
  counts: dict[str, int] = {}
129
130
  for trace in scored:
130
- for name, value in {**trace.rewards, **trace.metrics}.items():
131
+ scores = {name: reward.score for name, reward in trace.rewards.items()}
132
+ for name, value in {**scores, **trace.metrics}.items():
131
133
  sums[name] = sums.get(name, 0.0) + value
132
134
  counts[name] = counts.get(name, 0) + 1
133
135
  n = len(scored)
verifiers/v1/trace.py CHANGED
@@ -266,7 +266,7 @@ _NODE_DUMP_EXCLUDE: dict = {
266
266
  """Raw tensor fields kept on the msgpack wire but excluded from JSON records."""
267
267
 
268
268
 
269
- TRACE_VERSION = 3
269
+ TRACE_VERSION = 4
270
270
  """Version of the trace record schema (see `Trace.model_json_schema()`). Bumped on
271
271
  breaking shape changes; optional-with-default fields are additive and don't bump it."""
272
272
 
@@ -343,6 +343,21 @@ class TraceTask(StrictBaseModel, Generic[DataT]):
343
343
  """The (immutable) row being solved."""
344
344
 
345
345
 
346
+ class Reward(StrictBaseModel):
347
+ """One named reward as recorded on the trace: the raw score next to its weight,
348
+ so records keep both readable and the weighted sum stays a derived view."""
349
+
350
+ score: float
351
+ """The raw value the reward function returned, unweighted."""
352
+ weight: float = 1.0
353
+ """The multiplier `score` carries in the trace-level `reward` sum."""
354
+
355
+ @property
356
+ def value(self) -> float:
357
+ """This reward's weighted contribution to the trace-level `reward`."""
358
+ return self.score * self.weight
359
+
360
+
346
361
  class Trace(StrictBaseModel, Generic[DataT, StateT, AgentConfigT]):
347
362
  id: str = Field(default_factory=lambda: uuid.uuid4().hex)
348
363
  """Unique id for this rollout, auto-generated per trace."""
@@ -369,8 +384,9 @@ class Trace(StrictBaseModel, Generic[DataT, StateT, AgentConfigT]):
369
384
  """Every provider exchange behind the sampled turns, in order: raw wire request/response
370
385
  plus per-call timing and errors, linked into `nodes` via `ModelCall.node`."""
371
386
 
372
- rewards: dict[str, float] = Field(default_factory=dict)
373
- """Weighted contributions from task rewards, judges, and the env's `score()`."""
387
+ rewards: dict[str, Reward] = Field(default_factory=dict)
388
+ """Named rewards from tasks, judges, and the env's `score()` — each keeps its
389
+ raw `score` and `weight`; the trace-level `reward` is their weighted sum."""
374
390
  metrics: dict[str, float] = Field(default_factory=dict)
375
391
  """Unweighted metrics from tasks, harnesses, and judges."""
376
392
  info: dict[str, Any] = Field(default_factory=dict)
@@ -400,7 +416,7 @@ class Trace(StrictBaseModel, Generic[DataT, StateT, AgentConfigT]):
400
416
 
401
417
  @property
402
418
  def reward(self) -> float:
403
- return sum(self.rewards.values())
419
+ return sum(r.value for r in self.rewards.values())
404
420
 
405
421
  @property
406
422
  def error(self) -> Error | None:
@@ -559,12 +575,12 @@ class Trace(StrictBaseModel, Generic[DataT, StateT, AgentConfigT]):
559
575
  self.extra_usage.append(response.usage)
560
576
 
561
577
  def record_reward(self, name: str, value: float, weight: float = 1.0) -> None:
562
- contribution = float(value) * float(weight)
578
+ reward = Reward(score=float(value), weight=float(weight))
563
579
  if name in self.rewards:
564
580
  logger.warning(
565
- "reward %r overridden: %s -> %s", name, self.rewards[name], contribution
581
+ "reward %r overridden: %s -> %s", name, self.rewards[name], reward
566
582
  )
567
- self.rewards[name] = contribution
583
+ self.rewards[name] = reward
568
584
 
569
585
  def stamp(self, run: RunInfo | None = None, **info: Any) -> None:
570
586
  """Stamp identity only the consumer knows (the eval CLI / a trainer) onto the
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: verifiers
3
- Version: 0.2.2.dev25
3
+ Version: 0.2.2.dev27
4
4
  Summary: Verifiers: Environments for LLM Reinforcement Learning
5
5
  Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
6
6
  Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
@@ -167,7 +167,7 @@ verifiers/utils/threaded_sandbox_client.py,sha256=Pbr8MA4FDPEitL6z88S8T1qLJOmtXN
167
167
  verifiers/utils/tool_utils.py,sha256=gInWZODWQUZN4TyEMuuITyOUm9qlHJxtcvr1zZl_zJs,1020
168
168
  verifiers/utils/usage_utils.py,sha256=GPLC0xGY_Obrr8X7huWY-2iODZ7tgX-MtO8WSDN4rXI,3904
169
169
  verifiers/utils/version_utils.py,sha256=am3hZLnlaUWTFdllGZR1VMN91hEhiiE8FDmPf1cOG1k,2642
170
- verifiers/v1/__init__.py,sha256=29oq_Ng33Shgo2x7OGciV7SQrN0ndB8eBxsbebCggqw,6833
170
+ verifiers/v1/__init__.py,sha256=z0yUgI1iVf8znsAtBG2fpRKuwipaQ5yS4PvGhWJbKtE,6859
171
171
  verifiers/v1/agent.py,sha256=FP38GsZXuEJe3tJ-U2THpwBf9NWQeo847hWRZoeggNw,31956
172
172
  verifiers/v1/decorators.py,sha256=XRMkUQSyvXCYP5fOwzBYV5qEOlxLg6rzYhnhrHHHTIQ,3838
173
173
  verifiers/v1/env.py,sha256=w6sHWdWijLRZzhwVyJjadnGtziNqJAPCeMUoNdzHoLg,17712
@@ -176,9 +176,9 @@ verifiers/v1/errors.py,sha256=Pj5Om8x1fP3TDPQQn6nUCoSPoZRsy2JeBz8pXhpPrDY,6883
176
176
  verifiers/v1/graph.py,sha256=Mivq5jICUcyK-fhomFTwP4fQn688MXg6-hmdf03aC4Y,27703
177
177
  verifiers/v1/harness.py,sha256=Ukzhk7wxSUSVnN4tRAUMfniN9R9MBSNzs2SiT5YcRzQ,10711
178
178
  verifiers/v1/judge.py,sha256=ZWvr6uCniSwqzzjKrlyh-lzfCh5q-VyB16W42x0aWlg,9449
179
- verifiers/v1/legacy.py,sha256=b53EDDOImu8rSaZR_ZE7jNaQTJ6aIw5T8hxlXU4fTh0,22280
179
+ verifiers/v1/legacy.py,sha256=8eVGhutQEgJG4qabhph4Xs1VzTWvmMxMn3-Zk-6ViWM,22306
180
180
  verifiers/v1/loaders.py,sha256=FdICekH_c9WYkUe0XJCI9AWxL1sJjjzzJPJ7ZJdeyRk,9962
181
- verifiers/v1/push.py,sha256=N8UNz0eHKVH_vgh6ZuN-efqPOgYtscQBzn8zz-JYmYw,11138
181
+ verifiers/v1/push.py,sha256=VfLESKAlL8WXzfXx6CrqwMJ-CvPTkP-tUf6Rin12V8w,11273
182
182
  verifiers/v1/retries.py,sha256=ZQxY6R_FoXooERmhIMZnhY3b2koJOVjaIvkSO9gw53E,5379
183
183
  verifiers/v1/rollout.py,sha256=8_LC938Ws9uWFRPz2DDpivttqDALb3Yir7J5wjnsoIE,20332
184
184
  verifiers/v1/scoring.py,sha256=I_mtqhTZ195hj2-CB5jzr3BY5GFKgNEvTf331f7Is4k,5740
@@ -186,7 +186,7 @@ verifiers/v1/session.py,sha256=J6ZPuvsg6vMTJAs9dMUMtfdjCuvp6Hom877DVdKFKlY,6912
186
186
  verifiers/v1/state.py,sha256=R8tyQv8nsFV2pztrquDqCOa1Mk1fAp3w2GjqUugZwE0,689
187
187
  verifiers/v1/task.py,sha256=qKAAF154ykgvrms2Y2Or6kgGVPvBMsFCD6KqX2QOQaU,10912
188
188
  verifiers/v1/taskset.py,sha256=PX1-skAVhSpGF07qjxcG0Ctmq1LiMftMxxok-uc8qRY,3939
189
- verifiers/v1/trace.py,sha256=J_g3DgruE8B747fjkmhiI9UE8KIesf9mqAANMNo6Z1E,25976
189
+ verifiers/v1/trace.py,sha256=OE7vW6sYGjA18JdRJuHM9UiRaq9XkGS0Yv2m7nuvEYs,26619
190
190
  verifiers/v1/types.py,sha256=5tZyG4r4bLJ9a14oybHy7bSCpt5X5A17eRwmIszuRo8,9118
191
191
  verifiers/v1/acp/__init__.py,sha256=9dwH6fLopNndRmcpRSYC8ghzzMquaLfgh645QM-Dwic,2189
192
192
  verifiers/v1/acp/_runner.py,sha256=FqEnHR0Q_YrCt_EuaJqiqTpd2_kFa41FqcEunwOhvTI,6856
@@ -201,7 +201,7 @@ verifiers/v1/cli/serve.py,sha256=VHzcr2bM8R3XoGEu6WvY1NSQoiBZsk49cVW5AB9N4Kg,266
201
201
  verifiers/v1/cli/validate.py,sha256=r3ByzcLc_DFkvlCB7NanYRnBSOIi5EU9cUANy74l3T0,9547
202
202
  verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
203
203
  verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
204
- verifiers/v1/cli/dashboard/eval.py,sha256=F--kqnwjeihY1IrQ55liMwpQJSIkcp1RZmZzhKLhXVY,34081
204
+ verifiers/v1/cli/dashboard/eval.py,sha256=nmaWIq-4ormHZWjwlWm1QSFo383Sc4VFEkw9rYnpi00,34357
205
205
  verifiers/v1/cli/dashboard/replay.py,sha256=CvRVaf0dUum5v0IzPQbFPPikBQmTrkMeaunGfVMTWHc,2755
206
206
  verifiers/v1/cli/dashboard/validate.py,sha256=rLsQ_31DjTIZpC_VO7zSG41ncUEBxqCx8mL9nTnGiMc,3650
207
207
  verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
@@ -235,8 +235,8 @@ verifiers/v1/dialects/base.py,sha256=YZQDnasQseDXK-GHLfJLZgyHIIygS80YhD_sTjmUbfA
235
235
  verifiers/v1/dialects/chat.py,sha256=DFvjmIW86jZUMWuz-hOqzxC6tiiFibiAzNEJ3JkpvjU,13998
236
236
  verifiers/v1/dialects/responses.py,sha256=vVgeT7-aWjtfED2IKlDwTATKn_AQZruHNL6mkFTTZTU,13548
237
237
  verifiers/v1/envs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
238
- verifiers/v1/envs/agentic_judge/__init__.py,sha256=p43dmdpO484Euu9urdUQ1kNaX_hJOTC2YDVcjPpjmQs,143
239
- verifiers/v1/envs/agentic_judge/env.py,sha256=jzoKZQzX8yFKomQabbx02YmA0et_1GybOHJIYnAhS6o,7523
238
+ verifiers/v1/envs/agentic_judge/__init__.py,sha256=X7vQbbbfmJ_j-mJEfBkuB4a8QUPkUYkpG7yp0xYOEmg,279
239
+ verifiers/v1/envs/agentic_judge/env.py,sha256=_zEKG9i3LsOVIQWULXYAHpbmrQNPWBRayJx2yjvtPbQ,15307
240
240
  verifiers/v1/envs/best_of_n/__init__.py,sha256=mubASiwXMNIhnQFDccNAc7yZ0OFsOVK8hH8WIx9b2A4,119
241
241
  verifiers/v1/envs/best_of_n/env.py,sha256=KEUJBCVqblnXP_lEwm3mjGdaikBMkecz8UjOFPBhxc0,1987
242
242
  verifiers/v1/envs/single_agent/__init__.py,sha256=r0edAc_6gtHBhPzeBGVcdHZZqSdCPZnLMje6eVgyz5U,138
@@ -327,8 +327,8 @@ verifiers/v1/utils/logging.py,sha256=OcMHA6NsYux3oIzjPuI95rDWmFBHNcHDjZeNIhXTX-Y
327
327
  verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
328
328
  verifiers/v1/utils/sampling.py,sha256=JczGzBn6s3wsIrSY2Hy3hE8m6NreNgsNzhSjDikQqX0,1037
329
329
  verifiers/v1/utils/version.py,sha256=75ZtI2NHBmlb52KpcKLUpiSXp8q4bASr7uKeXoCKlT8,1582
330
- verifiers-0.2.2.dev25.dist-info/METADATA,sha256=VBIg7f-noyBD7tjlgGM2pvAlM2SKGcvJRycLWfPXarA,4540
331
- verifiers-0.2.2.dev25.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
332
- verifiers-0.2.2.dev25.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
333
- verifiers-0.2.2.dev25.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
334
- verifiers-0.2.2.dev25.dist-info/RECORD,,
330
+ verifiers-0.2.2.dev27.dist-info/METADATA,sha256=JnRcvyGkm0iogzBO_JnbUyqr6DO1_EUfDXi-TCV6ezg,4540
331
+ verifiers-0.2.2.dev27.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
332
+ verifiers-0.2.2.dev27.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
333
+ verifiers-0.2.2.dev27.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
334
+ verifiers-0.2.2.dev27.dist-info/RECORD,,