verifiers 0.2.2.dev85__py3-none-any.whl → 0.2.2.dev86__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,16 +1,16 @@
1
1
  from verifiers.v1.envs.agentic_judge.env import (
2
- AgenticJudgeEnv,
3
2
  AgenticJudgeEnvConfig,
4
3
  Criterion,
4
+ IsolatedAgenticJudgeEnv,
5
5
  JudgeTaskConfig,
6
6
  ScoreConfig,
7
7
  TextFile,
8
8
  )
9
9
 
10
10
  __all__ = [
11
- "AgenticJudgeEnv",
12
11
  "AgenticJudgeEnvConfig",
13
12
  "Criterion",
13
+ "IsolatedAgenticJudgeEnv",
14
14
  "JudgeTaskConfig",
15
15
  "ScoreConfig",
16
16
  "TextFile",
@@ -1,16 +1,17 @@
1
- """agentic-judge: a solver plays the task, then a judge verifies the work.
2
-
3
- A reusable env (`--env.id agentic-judge` over any taskset). The solver plays the
4
- task in a container provisioned from its runtime policy; the judge then grades
5
- rubric criteria (`[env.task]`: policy prompt, criteria file) and writes its
6
- verdicts to `/tmp/verdict.json`, with the solver's full trace record uploaded at
7
- `/tmp/trace.json`. `finalize()` validates them strictly onto the solver's trace —
8
- `judge/<name>` metrics plus a weighted-mean `judge` reward, composed with the
9
- taskset's own rewards via `[env.score]` (judge-only by default).
10
-
11
- `--env.share-runtime` controls whether the judge uses the solver's runtime. It is
12
- enabled by default. When disabled, the judge gets a fresh runtime containing the
13
- task's collected artifacts.
1
+ """Agentic judging: a solver plays the task, then a judge verifies the work.
2
+
3
+ Two reusable envs share the grading protocol. `--env.id agentic-judge` provisions
4
+ a fresh box from the solver's runtime policy and restores only the task's collected
5
+ artifacts; `--env.id shared-agentic-judge` explicitly runs the judge in the
6
+ solver's box. The judge grades rubric criteria (`[env.task]`: policy prompt,
7
+ criteria file) and writes its verdicts to `/tmp/verdict.json`, with the solver's
8
+ full trace record uploaded at `/tmp/trace.json`. `finalize()` validates them
9
+ strictly onto the solver's trace — `judge/<name>` metrics plus a weighted-mean
10
+ `judge` reward, composed with the taskset's own rewards via `[env.score]`
11
+ (judge-only by default).
12
+
13
+ The environment id selects the runtime boundary; there is no mode boolean whose
14
+ value can disagree with the environment's security and artifact semantics.
14
15
  """
15
16
 
16
17
  import json
@@ -260,20 +261,22 @@ class AgenticJudgeEnvConfig(vf.EnvConfig):
260
261
  """The solver agent. Its runtime must be a container:
261
262
  `--env.solver.runtime.type docker|prime`."""
262
263
  judge: vf.AgentConfig = vf.AgentConfig()
263
- """The judge agent. Its runtime is ignored when `share_runtime` is enabled;
264
- otherwise it must be a container."""
265
- share_runtime: bool = True
266
- """Whether the judge grades in the solver's runtime."""
264
+ """The judge agent. Its runtime setting is ignored: both judging modes use the
265
+ solver's resolved runtime policy, either by borrowing its box or provisioning a
266
+ fresh equivalent one."""
267
267
  task: JudgeTaskConfig = JudgeTaskConfig()
268
268
  score: ScoreConfig = ScoreConfig()
269
269
 
270
270
 
271
271
  class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
272
+ """Common agentic-judge protocol; subclasses choose the runtime boundary."""
273
+
272
274
  def __init__(self, config: AgenticJudgeEnvConfig) -> None:
273
- if config.share_runtime:
274
- config.judge = config.judge.model_copy(
275
- update={"runtime": config.solver.runtime}
276
- )
275
+ # Both modes use the solver's policy. Shared judging borrows that exact box;
276
+ # isolated judging resolves the mirrored JudgeTask into a fresh equivalent.
277
+ config.judge = config.judge.model_copy(
278
+ update={"runtime": config.solver.runtime}
279
+ )
277
280
  super().__init__(config)
278
281
  self._check_agents()
279
282
  # A missing policy file or a malformed rubric fails here, not mid-episode.
@@ -304,21 +307,6 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
304
307
  # The judge grades the policy; its tokens are never training data.
305
308
  agents.judge.trainable = False
306
309
 
307
- async def run(self, task: vf.Task, agents: vf.Agents) -> None:
308
- if self.config.share_runtime:
309
- async with agents.solver.provision(task) as box:
310
- solution = await agents.solver.run(task, runtime=box)
311
- judge_task = JudgeTask.from_trace(solution, self.config.task)
312
- await agents.judge.run(judge_task, runtime=box)
313
- return
314
-
315
- solution = await agents.solver.run(task)
316
- if not solution.ok:
317
- return
318
- await agents.judge.run(
319
- JudgeTask.from_trace(solution, self.config.task, share_runtime=False)
320
- )
321
-
322
310
  async def finalize(self, task: vf.Task, episode: vf.Episode) -> None:
323
311
  by_agent = {t.agent.name: t for t in episode.traces}
324
312
  if "judge" not in by_agent:
@@ -336,3 +324,25 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
336
324
  total = sum(criterion.weight for criterion in criteria)
337
325
  reward = sum(c.weight * scores[c.name] for c in criteria) / total
338
326
  solution.record_reward("judge", reward, weight=self.config.score.judge_weight)
327
+
328
+
329
+ class SharedAgenticJudgeEnv(AgenticJudgeEnv):
330
+ """Judge the solver in its runtime, preserving the complete mutable workspace."""
331
+
332
+ async def run(self, task: vf.Task, agents: vf.Agents) -> None:
333
+ async with agents.solver.provision(task) as box:
334
+ solution = await agents.solver.run(task, runtime=box)
335
+ judge_task = JudgeTask.from_trace(solution, self.config.task)
336
+ await agents.judge.run(judge_task, runtime=box)
337
+
338
+
339
+ class IsolatedAgenticJudgeEnv(AgenticJudgeEnv):
340
+ """Judge only collected artifacts in a fresh box with the solver's policy."""
341
+
342
+ async def run(self, task: vf.Task, agents: vf.Agents) -> None:
343
+ solution = await agents.solver.run(task)
344
+ if not solution.ok:
345
+ return
346
+ await agents.judge.run(
347
+ JudgeTask.from_trace(solution, self.config.task, share_runtime=False)
348
+ )
@@ -0,0 +1,6 @@
1
+ from verifiers.v1.envs.agentic_judge.env import (
2
+ AgenticJudgeEnvConfig,
3
+ SharedAgenticJudgeEnv,
4
+ )
5
+
6
+ __all__ = ["AgenticJudgeEnvConfig", "SharedAgenticJudgeEnv"]
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: verifiers
3
- Version: 0.2.2.dev85
3
+ Version: 0.2.2.dev86
4
4
  Summary: Verifiers: Environments for LLM Reinforcement Learning
5
5
  Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
6
6
  Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
@@ -231,10 +231,11 @@ verifiers/v1/dialects/base.py,sha256=YZQDnasQseDXK-GHLfJLZgyHIIygS80YhD_sTjmUbfA
231
231
  verifiers/v1/dialects/chat.py,sha256=R_otqi5N0snB1QVKIYloNeLHR9kc7OmI06HYg1ir_Lw,14021
232
232
  verifiers/v1/dialects/responses.py,sha256=ismnxUikPcPUIm80FUTPBbj-Cn1e0yGAsISMEseAWRc,13679
233
233
  verifiers/v1/envs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
234
- verifiers/v1/envs/agentic_judge/__init__.py,sha256=6vwtMCQ_cuvuubCF5-nrE5W7gGgzE-LIPKorkM5DXdw,309
235
- verifiers/v1/envs/agentic_judge/env.py,sha256=ZgNDanhSQYpP3K5Vz2rUW0UjWGApdqz86ng9BXhj6PA,13914
234
+ verifiers/v1/envs/agentic_judge/__init__.py,sha256=LfEcRHLG_siKzhzWmODDXWSLpLgfum3UFdQDK9XG0gk,325
235
+ verifiers/v1/envs/agentic_judge/env.py,sha256=9HdI8dm6m5QTNcHhpMvu6JJ2GlqH6e6xrc0inrEEZMk,14443
236
236
  verifiers/v1/envs/best_of_n/__init__.py,sha256=mubASiwXMNIhnQFDccNAc7yZ0OFsOVK8hH8WIx9b2A4,119
237
237
  verifiers/v1/envs/best_of_n/env.py,sha256=KEUJBCVqblnXP_lEwm3mjGdaikBMkecz8UjOFPBhxc0,1987
238
+ verifiers/v1/envs/shared_agentic_judge/__init__.py,sha256=9T4vyYAuqgZSQjqTIceghxnpauvqeW6XGhZSQDCttMA,168
238
239
  verifiers/v1/envs/single_agent/__init__.py,sha256=r0edAc_6gtHBhPzeBGVcdHZZqSdCPZnLMje6eVgyz5U,138
239
240
  verifiers/v1/envs/single_agent/env.py,sha256=lPs_a6VA7KoH1xqA3T8S32Jzb_d4zrx0_54LrqT5z1Q,1187
240
241
  verifiers/v1/envs/user_sim/__init__.py,sha256=nnT695HdqjAEouYL5wtVu6MhIp2wf5yuK-EYgxOVx0s,118
@@ -337,8 +338,8 @@ verifiers/v1/utils/platform.py,sha256=56Ixmk1SER6q5LvzyYcA0hgzGpzhxKZL8Hiqp4c_XV
337
338
  verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
338
339
  verifiers/v1/utils/score.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
339
340
  verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
340
- verifiers-0.2.2.dev85.dist-info/METADATA,sha256=SXrfQrFds4Ppx3ye69DlhIspd3t9JAz7gx4EfYB7CA0,4545
341
- verifiers-0.2.2.dev85.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
342
- verifiers-0.2.2.dev85.dist-info/entry_points.txt,sha256=v6v0QT9vVExnfn4br42MostI_o3f3ZbWYlTormz-U3g,515
343
- verifiers-0.2.2.dev85.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
344
- verifiers-0.2.2.dev85.dist-info/RECORD,,
341
+ verifiers-0.2.2.dev86.dist-info/METADATA,sha256=OcKmRg99ZpwQgZq56d-TeFt-IOtk-Jv3EUv2x_OvBcI,4545
342
+ verifiers-0.2.2.dev86.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
343
+ verifiers-0.2.2.dev86.dist-info/entry_points.txt,sha256=v6v0QT9vVExnfn4br42MostI_o3f3ZbWYlTormz-U3g,515
344
+ verifiers-0.2.2.dev86.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
345
+ verifiers-0.2.2.dev86.dist-info/RECORD,,