verifiers 0.2.2.dev84__py3-none-any.whl → 0.2.2.dev86__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/envs/agentic_judge/__init__.py +2 -2
- verifiers/v1/envs/agentic_judge/env.py +48 -36
- verifiers/v1/envs/shared_agentic_judge/__init__.py +6 -0
- verifiers/v1/runtimes/base.py +5 -7
- verifiers/v1/runtimes/docker/__init__.py +8 -1
- verifiers/v1/runtimes/prime.py +16 -13
- {verifiers-0.2.2.dev84.dist-info → verifiers-0.2.2.dev86.dist-info}/METADATA +1 -1
- {verifiers-0.2.2.dev84.dist-info → verifiers-0.2.2.dev86.dist-info}/RECORD +11 -10
- {verifiers-0.2.2.dev84.dist-info → verifiers-0.2.2.dev86.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev84.dist-info → verifiers-0.2.2.dev86.dist-info}/entry_points.txt +0 -0
- {verifiers-0.2.2.dev84.dist-info → verifiers-0.2.2.dev86.dist-info}/licenses/LICENSE +0 -0
|
@@ -1,16 +1,16 @@
|
|
|
1
1
|
from verifiers.v1.envs.agentic_judge.env import (
|
|
2
|
-
AgenticJudgeEnv,
|
|
3
2
|
AgenticJudgeEnvConfig,
|
|
4
3
|
Criterion,
|
|
4
|
+
IsolatedAgenticJudgeEnv,
|
|
5
5
|
JudgeTaskConfig,
|
|
6
6
|
ScoreConfig,
|
|
7
7
|
TextFile,
|
|
8
8
|
)
|
|
9
9
|
|
|
10
10
|
__all__ = [
|
|
11
|
-
"AgenticJudgeEnv",
|
|
12
11
|
"AgenticJudgeEnvConfig",
|
|
13
12
|
"Criterion",
|
|
13
|
+
"IsolatedAgenticJudgeEnv",
|
|
14
14
|
"JudgeTaskConfig",
|
|
15
15
|
"ScoreConfig",
|
|
16
16
|
"TextFile",
|
|
@@ -1,16 +1,17 @@
|
|
|
1
|
-
"""
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
`/tmp/
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
1
|
+
"""Agentic judging: a solver plays the task, then a judge verifies the work.
|
|
2
|
+
|
|
3
|
+
Two reusable envs share the grading protocol. `--env.id agentic-judge` provisions
|
|
4
|
+
a fresh box from the solver's runtime policy and restores only the task's collected
|
|
5
|
+
artifacts; `--env.id shared-agentic-judge` explicitly runs the judge in the
|
|
6
|
+
solver's box. The judge grades rubric criteria (`[env.task]`: policy prompt,
|
|
7
|
+
criteria file) and writes its verdicts to `/tmp/verdict.json`, with the solver's
|
|
8
|
+
full trace record uploaded at `/tmp/trace.json`. `finalize()` validates them
|
|
9
|
+
strictly onto the solver's trace — `judge/<name>` metrics plus a weighted-mean
|
|
10
|
+
`judge` reward, composed with the taskset's own rewards via `[env.score]`
|
|
11
|
+
(judge-only by default).
|
|
12
|
+
|
|
13
|
+
The environment id selects the runtime boundary; there is no mode boolean whose
|
|
14
|
+
value can disagree with the environment's security and artifact semantics.
|
|
14
15
|
"""
|
|
15
16
|
|
|
16
17
|
import json
|
|
@@ -165,6 +166,8 @@ class JudgeTask(vf.Task):
|
|
|
165
166
|
image=solved.image,
|
|
166
167
|
workdir=solved.workdir,
|
|
167
168
|
resources=solved.resources,
|
|
169
|
+
network_allow=solved.network_allow,
|
|
170
|
+
network_block=solved.network_block,
|
|
168
171
|
),
|
|
169
172
|
files=files,
|
|
170
173
|
artifacts={} if share_runtime else solution.state.artifacts,
|
|
@@ -258,20 +261,22 @@ class AgenticJudgeEnvConfig(vf.EnvConfig):
|
|
|
258
261
|
"""The solver agent. Its runtime must be a container:
|
|
259
262
|
`--env.solver.runtime.type docker|prime`."""
|
|
260
263
|
judge: vf.AgentConfig = vf.AgentConfig()
|
|
261
|
-
"""The judge agent. Its runtime is ignored
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
"""Whether the judge grades in the solver's runtime."""
|
|
264
|
+
"""The judge agent. Its runtime setting is ignored: both judging modes use the
|
|
265
|
+
solver's resolved runtime policy, either by borrowing its box or provisioning a
|
|
266
|
+
fresh equivalent one."""
|
|
265
267
|
task: JudgeTaskConfig = JudgeTaskConfig()
|
|
266
268
|
score: ScoreConfig = ScoreConfig()
|
|
267
269
|
|
|
268
270
|
|
|
269
271
|
class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
|
|
272
|
+
"""Common agentic-judge protocol; subclasses choose the runtime boundary."""
|
|
273
|
+
|
|
270
274
|
def __init__(self, config: AgenticJudgeEnvConfig) -> None:
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
+
# Both modes use the solver's policy. Shared judging borrows that exact box;
|
|
276
|
+
# isolated judging resolves the mirrored JudgeTask into a fresh equivalent.
|
|
277
|
+
config.judge = config.judge.model_copy(
|
|
278
|
+
update={"runtime": config.solver.runtime}
|
|
279
|
+
)
|
|
275
280
|
super().__init__(config)
|
|
276
281
|
self._check_agents()
|
|
277
282
|
# A missing policy file or a malformed rubric fails here, not mid-episode.
|
|
@@ -302,21 +307,6 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
|
|
|
302
307
|
# The judge grades the policy; its tokens are never training data.
|
|
303
308
|
agents.judge.trainable = False
|
|
304
309
|
|
|
305
|
-
async def run(self, task: vf.Task, agents: vf.Agents) -> None:
|
|
306
|
-
if self.config.share_runtime:
|
|
307
|
-
async with agents.solver.provision(task) as box:
|
|
308
|
-
solution = await agents.solver.run(task, runtime=box)
|
|
309
|
-
judge_task = JudgeTask.from_trace(solution, self.config.task)
|
|
310
|
-
await agents.judge.run(judge_task, runtime=box)
|
|
311
|
-
return
|
|
312
|
-
|
|
313
|
-
solution = await agents.solver.run(task)
|
|
314
|
-
if not solution.ok:
|
|
315
|
-
return
|
|
316
|
-
await agents.judge.run(
|
|
317
|
-
JudgeTask.from_trace(solution, self.config.task, share_runtime=False)
|
|
318
|
-
)
|
|
319
|
-
|
|
320
310
|
async def finalize(self, task: vf.Task, episode: vf.Episode) -> None:
|
|
321
311
|
by_agent = {t.agent.name: t for t in episode.traces}
|
|
322
312
|
if "judge" not in by_agent:
|
|
@@ -334,3 +324,25 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
|
|
|
334
324
|
total = sum(criterion.weight for criterion in criteria)
|
|
335
325
|
reward = sum(c.weight * scores[c.name] for c in criteria) / total
|
|
336
326
|
solution.record_reward("judge", reward, weight=self.config.score.judge_weight)
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
class SharedAgenticJudgeEnv(AgenticJudgeEnv):
|
|
330
|
+
"""Judge the solver in its runtime, preserving the complete mutable workspace."""
|
|
331
|
+
|
|
332
|
+
async def run(self, task: vf.Task, agents: vf.Agents) -> None:
|
|
333
|
+
async with agents.solver.provision(task) as box:
|
|
334
|
+
solution = await agents.solver.run(task, runtime=box)
|
|
335
|
+
judge_task = JudgeTask.from_trace(solution, self.config.task)
|
|
336
|
+
await agents.judge.run(judge_task, runtime=box)
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
class IsolatedAgenticJudgeEnv(AgenticJudgeEnv):
|
|
340
|
+
"""Judge only collected artifacts in a fresh box with the solver's policy."""
|
|
341
|
+
|
|
342
|
+
async def run(self, task: vf.Task, agents: vf.Agents) -> None:
|
|
343
|
+
solution = await agents.solver.run(task)
|
|
344
|
+
if not solution.ok:
|
|
345
|
+
return
|
|
346
|
+
await agents.judge.run(
|
|
347
|
+
JudgeTask.from_trace(solution, self.config.task, share_runtime=False)
|
|
348
|
+
)
|
verifiers/v1/runtimes/base.py
CHANGED
|
@@ -332,19 +332,17 @@ class Runtime(ABC):
|
|
|
332
332
|
return url
|
|
333
333
|
|
|
334
334
|
async def prepare_setup(self) -> None:
|
|
335
|
-
"""
|
|
335
|
+
"""Open egress for trusted setup when reusing a restricted runtime."""
|
|
336
336
|
if not self.network_restricted:
|
|
337
337
|
return
|
|
338
338
|
if self._setup_claimed:
|
|
339
|
-
|
|
340
|
-
f"network-filtered {self.type} runtimes are single-rollout; "
|
|
341
|
-
"provision a fresh runtime instead of reusing this one"
|
|
342
|
-
)
|
|
339
|
+
await self.prepare_execution(None)
|
|
343
340
|
self._setup_claimed = True
|
|
344
341
|
|
|
345
|
-
async def prepare_execution(self, routes: list[str]) -> None:
|
|
342
|
+
async def prepare_execution(self, routes: list[str] | None) -> None:
|
|
346
343
|
"""Last setup step, right before the agent starts. Restricted runtimes enforce
|
|
347
|
-
their policy here; `routes` identifies the interception and MCP endpoints.
|
|
344
|
+
their policy here; `routes` identifies the interception and MCP endpoints.
|
|
345
|
+
None restores unrestricted egress for another trusted setup phase."""
|
|
348
346
|
|
|
349
347
|
@property
|
|
350
348
|
def network_restricted(self) -> bool:
|
|
@@ -233,11 +233,16 @@ class DockerRuntime(Runtime):
|
|
|
233
233
|
return url.replace(host, "host.docker.internal", 1)
|
|
234
234
|
return url
|
|
235
235
|
|
|
236
|
-
async def prepare_execution(self, routes: list[str]) -> None:
|
|
236
|
+
async def prepare_execution(self, routes: list[str] | None) -> None:
|
|
237
237
|
"""Allow the declared framework routes, then leave the proxy as the only route."""
|
|
238
238
|
if not self.network_restricted:
|
|
239
239
|
return
|
|
240
240
|
assert self._proxy is not None
|
|
241
|
+
if routes is None:
|
|
242
|
+
self._proxy.policy = NetworkPolicy(
|
|
243
|
+
["*"], [], [HOST_ALIAS], allow_non_global=True
|
|
244
|
+
)
|
|
245
|
+
return
|
|
241
246
|
framework = [
|
|
242
247
|
urlsplit(url)._replace(path="", query="", fragment="").geturl()
|
|
243
248
|
for url in routes
|
|
@@ -247,6 +252,8 @@ class DockerRuntime(Runtime):
|
|
|
247
252
|
self.config.block,
|
|
248
253
|
framework,
|
|
249
254
|
)
|
|
255
|
+
if self._cut:
|
|
256
|
+
return
|
|
250
257
|
script = (
|
|
251
258
|
"set -eu; HOST=$1; "
|
|
252
259
|
"PORT=$2; "
|
verifiers/v1/runtimes/prime.py
CHANGED
|
@@ -181,22 +181,25 @@ class PrimeRuntime(Runtime):
|
|
|
181
181
|
) as e: # provisioning failure is one rollout's problem, not the eval's
|
|
182
182
|
raise SandboxError(f"prime sandbox provisioning failed: {e}") from e
|
|
183
183
|
|
|
184
|
-
async def prepare_execution(self, routes: list[str]) -> None:
|
|
184
|
+
async def prepare_execution(self, routes: list[str] | None) -> None:
|
|
185
185
|
"""Apply the host policy after setup and wait until the platform enforces it."""
|
|
186
186
|
if not self.network_restricted:
|
|
187
187
|
return
|
|
188
188
|
try:
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
h for h in (urlsplit(route).hostname for route in routes) if h
|
|
192
|
-
)
|
|
193
|
-
)
|
|
194
|
-
if self.config.allow == ["*"]:
|
|
195
|
-
policy = {"deny": self.config.block}
|
|
189
|
+
if routes is None:
|
|
190
|
+
policy = {"allow": ["*"]}
|
|
196
191
|
else:
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
192
|
+
hosts = list(
|
|
193
|
+
dict.fromkeys(
|
|
194
|
+
h for h in (urlsplit(route).hostname for route in routes) if h
|
|
195
|
+
)
|
|
196
|
+
)
|
|
197
|
+
if self.config.allow == ["*"]:
|
|
198
|
+
policy = {"deny": self.config.block}
|
|
199
|
+
else:
|
|
200
|
+
entries = list(dict.fromkeys([*hosts, *self.config.allow]))
|
|
201
|
+
validate_egress_lists(entries, None)
|
|
202
|
+
policy = {"allow": entries} if entries else {"deny": ["*"]}
|
|
200
203
|
status = await self._client.set_network(self.info.id, **policy)
|
|
201
204
|
try:
|
|
202
205
|
async with asyncio.timeout(60):
|
|
@@ -217,8 +220,8 @@ class PrimeRuntime(Runtime):
|
|
|
217
220
|
logger.info(
|
|
218
221
|
"prime: egress policy applied on sandbox %s (allow=%s block=%s)",
|
|
219
222
|
self.info.id,
|
|
220
|
-
|
|
221
|
-
|
|
223
|
+
policy.get("allow"),
|
|
224
|
+
policy.get("deny"),
|
|
222
225
|
)
|
|
223
226
|
|
|
224
227
|
async def run(self, argv: list[str], env: dict[str, str]) -> ProgramResult:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.dev86
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -231,10 +231,11 @@ verifiers/v1/dialects/base.py,sha256=YZQDnasQseDXK-GHLfJLZgyHIIygS80YhD_sTjmUbfA
|
|
|
231
231
|
verifiers/v1/dialects/chat.py,sha256=R_otqi5N0snB1QVKIYloNeLHR9kc7OmI06HYg1ir_Lw,14021
|
|
232
232
|
verifiers/v1/dialects/responses.py,sha256=ismnxUikPcPUIm80FUTPBbj-Cn1e0yGAsISMEseAWRc,13679
|
|
233
233
|
verifiers/v1/envs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
234
|
-
verifiers/v1/envs/agentic_judge/__init__.py,sha256=
|
|
235
|
-
verifiers/v1/envs/agentic_judge/env.py,sha256=
|
|
234
|
+
verifiers/v1/envs/agentic_judge/__init__.py,sha256=LfEcRHLG_siKzhzWmODDXWSLpLgfum3UFdQDK9XG0gk,325
|
|
235
|
+
verifiers/v1/envs/agentic_judge/env.py,sha256=9HdI8dm6m5QTNcHhpMvu6JJ2GlqH6e6xrc0inrEEZMk,14443
|
|
236
236
|
verifiers/v1/envs/best_of_n/__init__.py,sha256=mubASiwXMNIhnQFDccNAc7yZ0OFsOVK8hH8WIx9b2A4,119
|
|
237
237
|
verifiers/v1/envs/best_of_n/env.py,sha256=KEUJBCVqblnXP_lEwm3mjGdaikBMkecz8UjOFPBhxc0,1987
|
|
238
|
+
verifiers/v1/envs/shared_agentic_judge/__init__.py,sha256=9T4vyYAuqgZSQjqTIceghxnpauvqeW6XGhZSQDCttMA,168
|
|
238
239
|
verifiers/v1/envs/single_agent/__init__.py,sha256=r0edAc_6gtHBhPzeBGVcdHZZqSdCPZnLMje6eVgyz5U,138
|
|
239
240
|
verifiers/v1/envs/single_agent/env.py,sha256=lPs_a6VA7KoH1xqA3T8S32Jzb_d4zrx0_54LrqT5z1Q,1187
|
|
240
241
|
verifiers/v1/envs/user_sim/__init__.py,sha256=nnT695HdqjAEouYL5wtVu6MhIp2wf5yuK-EYgxOVx0s,118
|
|
@@ -296,12 +297,12 @@ verifiers/v1/mcp/launch.py,sha256=wBpyeiOO86l1H2lVMbLVSV9Lbp90XeyXqauuK8snfEU,20
|
|
|
296
297
|
verifiers/v1/mcp/server.py,sha256=dRBxytFgS6rv1BS6t5EgBd3xeF6U1nTyt0ArDCLk3VE,12301
|
|
297
298
|
verifiers/v1/mcp/toolset.py,sha256=T8Ioiq4wphxVOTHUDf28ZoQBM2ZYEevGnGgw8TFEMiA,1002
|
|
298
299
|
verifiers/v1/runtimes/__init__.py,sha256=Kwe612RjtGzrQhc19QdXc7CmZAnQLH-EdK_xqkKP3Tk,2629
|
|
299
|
-
verifiers/v1/runtimes/base.py,sha256=
|
|
300
|
+
verifiers/v1/runtimes/base.py,sha256=0vLTBATJplxVAgXRNqbUzX4J26p88f-r7JQmWsOugxg,16180
|
|
300
301
|
verifiers/v1/runtimes/limiters.py,sha256=ZjjYUFEt20yH8ggL5uU9w9U500AOB44qSUTQ7bYyh0c,2730
|
|
301
302
|
verifiers/v1/runtimes/modal.py,sha256=eFiT-Ovf_RwQZbSw3Dwv2EXeHV0d6uGvPswqKYKCw5k,8826
|
|
302
|
-
verifiers/v1/runtimes/prime.py,sha256=
|
|
303
|
+
verifiers/v1/runtimes/prime.py,sha256=gJf-Ap5N-9Ecp-GpCNlPta7Gc9C5--Bzbk3rbK-vBIQ,14901
|
|
303
304
|
verifiers/v1/runtimes/subprocess.py,sha256=ZrHRlTuTyCrcSKpiPD2ElcPpe97EvVsXJ8BfwUaNXiI,5988
|
|
304
|
-
verifiers/v1/runtimes/docker/__init__.py,sha256=
|
|
305
|
+
verifiers/v1/runtimes/docker/__init__.py,sha256=ObGNk7Uc5I4C3sumwIuD0Z30sO3FeDxl62hElfrGyWw,14775
|
|
305
306
|
verifiers/v1/runtimes/docker/egress.py,sha256=37UXgNX9gnoLxjMg4KZaJIGoApCAM6MK22eJGNoMi38,14514
|
|
306
307
|
verifiers/v1/serve/__init__.py,sha256=sr_zkDzAHnOcQm-QRm8xwJBmCMFD-xhKIFPaDZvMbUs,641
|
|
307
308
|
verifiers/v1/serve/client.py,sha256=CxpOwCn4ehuCXfqDJeV9zj7O0XC7MPcI-OZ1L_CSDlo,7057
|
|
@@ -337,8 +338,8 @@ verifiers/v1/utils/platform.py,sha256=56Ixmk1SER6q5LvzyYcA0hgzGpzhxKZL8Hiqp4c_XV
|
|
|
337
338
|
verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
|
|
338
339
|
verifiers/v1/utils/score.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
|
|
339
340
|
verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
|
|
340
|
-
verifiers-0.2.2.
|
|
341
|
-
verifiers-0.2.2.
|
|
342
|
-
verifiers-0.2.2.
|
|
343
|
-
verifiers-0.2.2.
|
|
344
|
-
verifiers-0.2.2.
|
|
341
|
+
verifiers-0.2.2.dev86.dist-info/METADATA,sha256=OcKmRg99ZpwQgZq56d-TeFt-IOtk-Jv3EUv2x_OvBcI,4545
|
|
342
|
+
verifiers-0.2.2.dev86.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
343
|
+
verifiers-0.2.2.dev86.dist-info/entry_points.txt,sha256=v6v0QT9vVExnfn4br42MostI_o3f3ZbWYlTormz-U3g,515
|
|
344
|
+
verifiers-0.2.2.dev86.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
345
|
+
verifiers-0.2.2.dev86.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|