verifiers 0.2.2.dev84__py3-none-any.whl → 0.2.2.dev86__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,16 +1,16 @@
1
1
  from verifiers.v1.envs.agentic_judge.env import (
2
- AgenticJudgeEnv,
3
2
  AgenticJudgeEnvConfig,
4
3
  Criterion,
4
+ IsolatedAgenticJudgeEnv,
5
5
  JudgeTaskConfig,
6
6
  ScoreConfig,
7
7
  TextFile,
8
8
  )
9
9
 
10
10
  __all__ = [
11
- "AgenticJudgeEnv",
12
11
  "AgenticJudgeEnvConfig",
13
12
  "Criterion",
13
+ "IsolatedAgenticJudgeEnv",
14
14
  "JudgeTaskConfig",
15
15
  "ScoreConfig",
16
16
  "TextFile",
@@ -1,16 +1,17 @@
1
- """agentic-judge: a solver plays the task, then a judge verifies the work.
2
-
3
- A reusable env (`--env.id agentic-judge` over any taskset). The solver plays the
4
- task in a container provisioned from its runtime policy; the judge then grades
5
- rubric criteria (`[env.task]`: policy prompt, criteria file) and writes its
6
- verdicts to `/tmp/verdict.json`, with the solver's full trace record uploaded at
7
- `/tmp/trace.json`. `finalize()` validates them strictly onto the solver's trace —
8
- `judge/<name>` metrics plus a weighted-mean `judge` reward, composed with the
9
- taskset's own rewards via `[env.score]` (judge-only by default).
10
-
11
- `--env.share-runtime` controls whether the judge uses the solver's runtime. It is
12
- enabled by default. When disabled, the judge gets a fresh runtime containing the
13
- task's collected artifacts.
1
+ """Agentic judging: a solver plays the task, then a judge verifies the work.
2
+
3
+ Two reusable envs share the grading protocol. `--env.id agentic-judge` provisions
4
+ a fresh box from the solver's runtime policy and restores only the task's collected
5
+ artifacts; `--env.id shared-agentic-judge` explicitly runs the judge in the
6
+ solver's box. The judge grades rubric criteria (`[env.task]`: policy prompt,
7
+ criteria file) and writes its verdicts to `/tmp/verdict.json`, with the solver's
8
+ full trace record uploaded at `/tmp/trace.json`. `finalize()` validates them
9
+ strictly onto the solver's trace — `judge/<name>` metrics plus a weighted-mean
10
+ `judge` reward, composed with the taskset's own rewards via `[env.score]`
11
+ (judge-only by default).
12
+
13
+ The environment id selects the runtime boundary; there is no mode boolean whose
14
+ value can disagree with the environment's security and artifact semantics.
14
15
  """
15
16
 
16
17
  import json
@@ -165,6 +166,8 @@ class JudgeTask(vf.Task):
165
166
  image=solved.image,
166
167
  workdir=solved.workdir,
167
168
  resources=solved.resources,
169
+ network_allow=solved.network_allow,
170
+ network_block=solved.network_block,
168
171
  ),
169
172
  files=files,
170
173
  artifacts={} if share_runtime else solution.state.artifacts,
@@ -258,20 +261,22 @@ class AgenticJudgeEnvConfig(vf.EnvConfig):
258
261
  """The solver agent. Its runtime must be a container:
259
262
  `--env.solver.runtime.type docker|prime`."""
260
263
  judge: vf.AgentConfig = vf.AgentConfig()
261
- """The judge agent. Its runtime is ignored when `share_runtime` is enabled;
262
- otherwise it must be a container."""
263
- share_runtime: bool = True
264
- """Whether the judge grades in the solver's runtime."""
264
+ """The judge agent. Its runtime setting is ignored: both judging modes use the
265
+ solver's resolved runtime policy, either by borrowing its box or provisioning a
266
+ fresh equivalent one."""
265
267
  task: JudgeTaskConfig = JudgeTaskConfig()
266
268
  score: ScoreConfig = ScoreConfig()
267
269
 
268
270
 
269
271
  class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
272
+ """Common agentic-judge protocol; subclasses choose the runtime boundary."""
273
+
270
274
  def __init__(self, config: AgenticJudgeEnvConfig) -> None:
271
- if config.share_runtime:
272
- config.judge = config.judge.model_copy(
273
- update={"runtime": config.solver.runtime}
274
- )
275
+ # Both modes use the solver's policy. Shared judging borrows that exact box;
276
+ # isolated judging resolves the mirrored JudgeTask into a fresh equivalent.
277
+ config.judge = config.judge.model_copy(
278
+ update={"runtime": config.solver.runtime}
279
+ )
275
280
  super().__init__(config)
276
281
  self._check_agents()
277
282
  # A missing policy file or a malformed rubric fails here, not mid-episode.
@@ -302,21 +307,6 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
302
307
  # The judge grades the policy; its tokens are never training data.
303
308
  agents.judge.trainable = False
304
309
 
305
- async def run(self, task: vf.Task, agents: vf.Agents) -> None:
306
- if self.config.share_runtime:
307
- async with agents.solver.provision(task) as box:
308
- solution = await agents.solver.run(task, runtime=box)
309
- judge_task = JudgeTask.from_trace(solution, self.config.task)
310
- await agents.judge.run(judge_task, runtime=box)
311
- return
312
-
313
- solution = await agents.solver.run(task)
314
- if not solution.ok:
315
- return
316
- await agents.judge.run(
317
- JudgeTask.from_trace(solution, self.config.task, share_runtime=False)
318
- )
319
-
320
310
  async def finalize(self, task: vf.Task, episode: vf.Episode) -> None:
321
311
  by_agent = {t.agent.name: t for t in episode.traces}
322
312
  if "judge" not in by_agent:
@@ -334,3 +324,25 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
334
324
  total = sum(criterion.weight for criterion in criteria)
335
325
  reward = sum(c.weight * scores[c.name] for c in criteria) / total
336
326
  solution.record_reward("judge", reward, weight=self.config.score.judge_weight)
327
+
328
+
329
+ class SharedAgenticJudgeEnv(AgenticJudgeEnv):
330
+ """Judge the solver in its runtime, preserving the complete mutable workspace."""
331
+
332
+ async def run(self, task: vf.Task, agents: vf.Agents) -> None:
333
+ async with agents.solver.provision(task) as box:
334
+ solution = await agents.solver.run(task, runtime=box)
335
+ judge_task = JudgeTask.from_trace(solution, self.config.task)
336
+ await agents.judge.run(judge_task, runtime=box)
337
+
338
+
339
+ class IsolatedAgenticJudgeEnv(AgenticJudgeEnv):
340
+ """Judge only collected artifacts in a fresh box with the solver's policy."""
341
+
342
+ async def run(self, task: vf.Task, agents: vf.Agents) -> None:
343
+ solution = await agents.solver.run(task)
344
+ if not solution.ok:
345
+ return
346
+ await agents.judge.run(
347
+ JudgeTask.from_trace(solution, self.config.task, share_runtime=False)
348
+ )
@@ -0,0 +1,6 @@
1
+ from verifiers.v1.envs.agentic_judge.env import (
2
+ AgenticJudgeEnvConfig,
3
+ SharedAgenticJudgeEnv,
4
+ )
5
+
6
+ __all__ = ["AgenticJudgeEnvConfig", "SharedAgenticJudgeEnv"]
@@ -332,19 +332,17 @@ class Runtime(ABC):
332
332
  return url
333
333
 
334
334
  async def prepare_setup(self) -> None:
335
- """Claim the runtime for trusted setup; restricted runtimes may reject reuse."""
335
+ """Open egress for trusted setup when reusing a restricted runtime."""
336
336
  if not self.network_restricted:
337
337
  return
338
338
  if self._setup_claimed:
339
- raise SandboxError(
340
- f"network-filtered {self.type} runtimes are single-rollout; "
341
- "provision a fresh runtime instead of reusing this one"
342
- )
339
+ await self.prepare_execution(None)
343
340
  self._setup_claimed = True
344
341
 
345
- async def prepare_execution(self, routes: list[str]) -> None:
342
+ async def prepare_execution(self, routes: list[str] | None) -> None:
346
343
  """Last setup step, right before the agent starts. Restricted runtimes enforce
347
- their policy here; `routes` identifies the interception and MCP endpoints."""
344
+ their policy here; `routes` identifies the interception and MCP endpoints.
345
+ None restores unrestricted egress for another trusted setup phase."""
348
346
 
349
347
  @property
350
348
  def network_restricted(self) -> bool:
@@ -233,11 +233,16 @@ class DockerRuntime(Runtime):
233
233
  return url.replace(host, "host.docker.internal", 1)
234
234
  return url
235
235
 
236
- async def prepare_execution(self, routes: list[str]) -> None:
236
+ async def prepare_execution(self, routes: list[str] | None) -> None:
237
237
  """Allow the declared framework routes, then leave the proxy as the only route."""
238
238
  if not self.network_restricted:
239
239
  return
240
240
  assert self._proxy is not None
241
+ if routes is None:
242
+ self._proxy.policy = NetworkPolicy(
243
+ ["*"], [], [HOST_ALIAS], allow_non_global=True
244
+ )
245
+ return
241
246
  framework = [
242
247
  urlsplit(url)._replace(path="", query="", fragment="").geturl()
243
248
  for url in routes
@@ -247,6 +252,8 @@ class DockerRuntime(Runtime):
247
252
  self.config.block,
248
253
  framework,
249
254
  )
255
+ if self._cut:
256
+ return
250
257
  script = (
251
258
  "set -eu; HOST=$1; "
252
259
  "PORT=$2; "
@@ -181,22 +181,25 @@ class PrimeRuntime(Runtime):
181
181
  ) as e: # provisioning failure is one rollout's problem, not the eval's
182
182
  raise SandboxError(f"prime sandbox provisioning failed: {e}") from e
183
183
 
184
- async def prepare_execution(self, routes: list[str]) -> None:
184
+ async def prepare_execution(self, routes: list[str] | None) -> None:
185
185
  """Apply the host policy after setup and wait until the platform enforces it."""
186
186
  if not self.network_restricted:
187
187
  return
188
188
  try:
189
- hosts = list(
190
- dict.fromkeys(
191
- h for h in (urlsplit(route).hostname for route in routes) if h
192
- )
193
- )
194
- if self.config.allow == ["*"]:
195
- policy = {"deny": self.config.block}
189
+ if routes is None:
190
+ policy = {"allow": ["*"]}
196
191
  else:
197
- entries = list(dict.fromkeys([*hosts, *self.config.allow]))
198
- validate_egress_lists(entries, None)
199
- policy = {"allow": entries} if entries else {"deny": ["*"]}
192
+ hosts = list(
193
+ dict.fromkeys(
194
+ h for h in (urlsplit(route).hostname for route in routes) if h
195
+ )
196
+ )
197
+ if self.config.allow == ["*"]:
198
+ policy = {"deny": self.config.block}
199
+ else:
200
+ entries = list(dict.fromkeys([*hosts, *self.config.allow]))
201
+ validate_egress_lists(entries, None)
202
+ policy = {"allow": entries} if entries else {"deny": ["*"]}
200
203
  status = await self._client.set_network(self.info.id, **policy)
201
204
  try:
202
205
  async with asyncio.timeout(60):
@@ -217,8 +220,8 @@ class PrimeRuntime(Runtime):
217
220
  logger.info(
218
221
  "prime: egress policy applied on sandbox %s (allow=%s block=%s)",
219
222
  self.info.id,
220
- self.config.allow,
221
- self.config.block,
223
+ policy.get("allow"),
224
+ policy.get("deny"),
222
225
  )
223
226
 
224
227
  async def run(self, argv: list[str], env: dict[str, str]) -> ProgramResult:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: verifiers
3
- Version: 0.2.2.dev84
3
+ Version: 0.2.2.dev86
4
4
  Summary: Verifiers: Environments for LLM Reinforcement Learning
5
5
  Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
6
6
  Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
@@ -231,10 +231,11 @@ verifiers/v1/dialects/base.py,sha256=YZQDnasQseDXK-GHLfJLZgyHIIygS80YhD_sTjmUbfA
231
231
  verifiers/v1/dialects/chat.py,sha256=R_otqi5N0snB1QVKIYloNeLHR9kc7OmI06HYg1ir_Lw,14021
232
232
  verifiers/v1/dialects/responses.py,sha256=ismnxUikPcPUIm80FUTPBbj-Cn1e0yGAsISMEseAWRc,13679
233
233
  verifiers/v1/envs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
234
- verifiers/v1/envs/agentic_judge/__init__.py,sha256=6vwtMCQ_cuvuubCF5-nrE5W7gGgzE-LIPKorkM5DXdw,309
235
- verifiers/v1/envs/agentic_judge/env.py,sha256=0KXuTEg4nd0jprPqFgQYFxf2JLdPVbUmljbgPOT-5Bg,13810
234
+ verifiers/v1/envs/agentic_judge/__init__.py,sha256=LfEcRHLG_siKzhzWmODDXWSLpLgfum3UFdQDK9XG0gk,325
235
+ verifiers/v1/envs/agentic_judge/env.py,sha256=9HdI8dm6m5QTNcHhpMvu6JJ2GlqH6e6xrc0inrEEZMk,14443
236
236
  verifiers/v1/envs/best_of_n/__init__.py,sha256=mubASiwXMNIhnQFDccNAc7yZ0OFsOVK8hH8WIx9b2A4,119
237
237
  verifiers/v1/envs/best_of_n/env.py,sha256=KEUJBCVqblnXP_lEwm3mjGdaikBMkecz8UjOFPBhxc0,1987
238
+ verifiers/v1/envs/shared_agentic_judge/__init__.py,sha256=9T4vyYAuqgZSQjqTIceghxnpauvqeW6XGhZSQDCttMA,168
238
239
  verifiers/v1/envs/single_agent/__init__.py,sha256=r0edAc_6gtHBhPzeBGVcdHZZqSdCPZnLMje6eVgyz5U,138
239
240
  verifiers/v1/envs/single_agent/env.py,sha256=lPs_a6VA7KoH1xqA3T8S32Jzb_d4zrx0_54LrqT5z1Q,1187
240
241
  verifiers/v1/envs/user_sim/__init__.py,sha256=nnT695HdqjAEouYL5wtVu6MhIp2wf5yuK-EYgxOVx0s,118
@@ -296,12 +297,12 @@ verifiers/v1/mcp/launch.py,sha256=wBpyeiOO86l1H2lVMbLVSV9Lbp90XeyXqauuK8snfEU,20
296
297
  verifiers/v1/mcp/server.py,sha256=dRBxytFgS6rv1BS6t5EgBd3xeF6U1nTyt0ArDCLk3VE,12301
297
298
  verifiers/v1/mcp/toolset.py,sha256=T8Ioiq4wphxVOTHUDf28ZoQBM2ZYEevGnGgw8TFEMiA,1002
298
299
  verifiers/v1/runtimes/__init__.py,sha256=Kwe612RjtGzrQhc19QdXc7CmZAnQLH-EdK_xqkKP3Tk,2629
299
- verifiers/v1/runtimes/base.py,sha256=tZRELSfZmi3l_1BsjabnNC3mEPfqNvUNevdnRvLPv6w,16257
300
+ verifiers/v1/runtimes/base.py,sha256=0vLTBATJplxVAgXRNqbUzX4J26p88f-r7JQmWsOugxg,16180
300
301
  verifiers/v1/runtimes/limiters.py,sha256=ZjjYUFEt20yH8ggL5uU9w9U500AOB44qSUTQ7bYyh0c,2730
301
302
  verifiers/v1/runtimes/modal.py,sha256=eFiT-Ovf_RwQZbSw3Dwv2EXeHV0d6uGvPswqKYKCw5k,8826
302
- verifiers/v1/runtimes/prime.py,sha256=kwvPqUI8D-GvcFwYRdpTHE1hXS0vfJGPlFOLgzm4Mvs,14756
303
+ verifiers/v1/runtimes/prime.py,sha256=gJf-Ap5N-9Ecp-GpCNlPta7Gc9C5--Bzbk3rbK-vBIQ,14901
303
304
  verifiers/v1/runtimes/subprocess.py,sha256=ZrHRlTuTyCrcSKpiPD2ElcPpe97EvVsXJ8BfwUaNXiI,5988
304
- verifiers/v1/runtimes/docker/__init__.py,sha256=vyEgf6XgAwjGgBicJchJvNgcDgYXrRvyQC4W52-23GU,14556
305
+ verifiers/v1/runtimes/docker/__init__.py,sha256=ObGNk7Uc5I4C3sumwIuD0Z30sO3FeDxl62hElfrGyWw,14775
305
306
  verifiers/v1/runtimes/docker/egress.py,sha256=37UXgNX9gnoLxjMg4KZaJIGoApCAM6MK22eJGNoMi38,14514
306
307
  verifiers/v1/serve/__init__.py,sha256=sr_zkDzAHnOcQm-QRm8xwJBmCMFD-xhKIFPaDZvMbUs,641
307
308
  verifiers/v1/serve/client.py,sha256=CxpOwCn4ehuCXfqDJeV9zj7O0XC7MPcI-OZ1L_CSDlo,7057
@@ -337,8 +338,8 @@ verifiers/v1/utils/platform.py,sha256=56Ixmk1SER6q5LvzyYcA0hgzGpzhxKZL8Hiqp4c_XV
337
338
  verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
338
339
  verifiers/v1/utils/score.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
339
340
  verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
340
- verifiers-0.2.2.dev84.dist-info/METADATA,sha256=hu_D0aS0L0VjJxeOvodUkjLmh1Ag4g4VU7iMaD8Yo2g,4545
341
- verifiers-0.2.2.dev84.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
342
- verifiers-0.2.2.dev84.dist-info/entry_points.txt,sha256=v6v0QT9vVExnfn4br42MostI_o3f3ZbWYlTormz-U3g,515
343
- verifiers-0.2.2.dev84.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
344
- verifiers-0.2.2.dev84.dist-info/RECORD,,
341
+ verifiers-0.2.2.dev86.dist-info/METADATA,sha256=OcKmRg99ZpwQgZq56d-TeFt-IOtk-Jv3EUv2x_OvBcI,4545
342
+ verifiers-0.2.2.dev86.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
343
+ verifiers-0.2.2.dev86.dist-info/entry_points.txt,sha256=v6v0QT9vVExnfn4br42MostI_o3f3ZbWYlTormz-U3g,515
344
+ verifiers-0.2.2.dev86.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
345
+ verifiers-0.2.2.dev86.dist-info/RECORD,,