verifiers 0.2.2.dev85__py3-none-any.whl → 0.2.2.dev87__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/configs/harness.py +2 -2
- verifiers/v1/envs/agentic_judge/__init__.py +2 -2
- verifiers/v1/envs/agentic_judge/env.py +46 -36
- verifiers/v1/envs/shared_agentic_judge/__init__.py +6 -0
- {verifiers-0.2.2.dev85.dist-info → verifiers-0.2.2.dev87.dist-info}/METADATA +1 -1
- {verifiers-0.2.2.dev85.dist-info → verifiers-0.2.2.dev87.dist-info}/RECORD +9 -8
- {verifiers-0.2.2.dev85.dist-info → verifiers-0.2.2.dev87.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev85.dist-info → verifiers-0.2.2.dev87.dist-info}/entry_points.txt +0 -0
- {verifiers-0.2.2.dev85.dist-info → verifiers-0.2.2.dev87.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/configs/harness.py
CHANGED
|
@@ -5,7 +5,7 @@ from __future__ import annotations
|
|
|
5
5
|
import os
|
|
6
6
|
from pathlib import Path
|
|
7
7
|
|
|
8
|
-
from pydantic import ConfigDict, Field
|
|
8
|
+
from pydantic import ConfigDict, Field, FiniteFloat
|
|
9
9
|
from pydantic_config import BaseConfig
|
|
10
10
|
|
|
11
11
|
from verifiers.v1.types import ID
|
|
@@ -20,7 +20,7 @@ class HarnessConfig(BaseConfig):
|
|
|
20
20
|
"""Extra program variables; harness-owned variables take precedence."""
|
|
21
21
|
forward_env: list[str] = Field(default_factory=list)
|
|
22
22
|
"""Host variables to forward without writing secrets into config; explicit `env` wins."""
|
|
23
|
-
tool_timeout:
|
|
23
|
+
tool_timeout: FiniteFloat = Field(600.0, gt=0)
|
|
24
24
|
"""Seconds a single MCP tool call may take; raise it for tools that boot a VM."""
|
|
25
25
|
disabled_tools: list[str] | None = None
|
|
26
26
|
skills: list[Path] = Field(default_factory=list)
|
|
@@ -1,16 +1,16 @@
|
|
|
1
1
|
from verifiers.v1.envs.agentic_judge.env import (
|
|
2
|
-
AgenticJudgeEnv,
|
|
3
2
|
AgenticJudgeEnvConfig,
|
|
4
3
|
Criterion,
|
|
4
|
+
IsolatedAgenticJudgeEnv,
|
|
5
5
|
JudgeTaskConfig,
|
|
6
6
|
ScoreConfig,
|
|
7
7
|
TextFile,
|
|
8
8
|
)
|
|
9
9
|
|
|
10
10
|
__all__ = [
|
|
11
|
-
"AgenticJudgeEnv",
|
|
12
11
|
"AgenticJudgeEnvConfig",
|
|
13
12
|
"Criterion",
|
|
13
|
+
"IsolatedAgenticJudgeEnv",
|
|
14
14
|
"JudgeTaskConfig",
|
|
15
15
|
"ScoreConfig",
|
|
16
16
|
"TextFile",
|
|
@@ -1,16 +1,17 @@
|
|
|
1
|
-
"""
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
`/tmp/
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
1
|
+
"""Agentic judging: a solver plays the task, then a judge verifies the work.
|
|
2
|
+
|
|
3
|
+
Two reusable envs share the grading protocol. `--env.id agentic-judge` provisions
|
|
4
|
+
a fresh box from the solver's runtime policy and restores only the task's collected
|
|
5
|
+
artifacts; `--env.id shared-agentic-judge` explicitly runs the judge in the
|
|
6
|
+
solver's box. The judge grades rubric criteria (`[env.task]`: policy prompt,
|
|
7
|
+
criteria file) and writes its verdicts to `/tmp/verdict.json`, with the solver's
|
|
8
|
+
full trace record uploaded at `/tmp/trace.json`. `finalize()` validates them
|
|
9
|
+
strictly onto the solver's trace — `judge/<name>` metrics plus a weighted-mean
|
|
10
|
+
`judge` reward, composed with the taskset's own rewards via `[env.score]`
|
|
11
|
+
(judge-only by default).
|
|
12
|
+
|
|
13
|
+
The environment id selects the runtime boundary; there is no mode boolean whose
|
|
14
|
+
value can disagree with the environment's security and artifact semantics.
|
|
14
15
|
"""
|
|
15
16
|
|
|
16
17
|
import json
|
|
@@ -260,20 +261,22 @@ class AgenticJudgeEnvConfig(vf.EnvConfig):
|
|
|
260
261
|
"""The solver agent. Its runtime must be a container:
|
|
261
262
|
`--env.solver.runtime.type docker|prime`."""
|
|
262
263
|
judge: vf.AgentConfig = vf.AgentConfig()
|
|
263
|
-
"""The judge agent. Its runtime is ignored
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
"""Whether the judge grades in the solver's runtime."""
|
|
264
|
+
"""The judge agent. Its runtime setting is ignored: both judging modes use the
|
|
265
|
+
solver's resolved runtime policy, either by borrowing its box or provisioning a
|
|
266
|
+
fresh equivalent one."""
|
|
267
267
|
task: JudgeTaskConfig = JudgeTaskConfig()
|
|
268
268
|
score: ScoreConfig = ScoreConfig()
|
|
269
269
|
|
|
270
270
|
|
|
271
271
|
class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
|
|
272
|
+
"""Common agentic-judge protocol; subclasses choose the runtime boundary."""
|
|
273
|
+
|
|
272
274
|
def __init__(self, config: AgenticJudgeEnvConfig) -> None:
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
275
|
+
# Both modes use the solver's policy. Shared judging borrows that exact box;
|
|
276
|
+
# isolated judging resolves the mirrored JudgeTask into a fresh equivalent.
|
|
277
|
+
config.judge = config.judge.model_copy(
|
|
278
|
+
update={"runtime": config.solver.runtime}
|
|
279
|
+
)
|
|
277
280
|
super().__init__(config)
|
|
278
281
|
self._check_agents()
|
|
279
282
|
# A missing policy file or a malformed rubric fails here, not mid-episode.
|
|
@@ -304,21 +307,6 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
|
|
|
304
307
|
# The judge grades the policy; its tokens are never training data.
|
|
305
308
|
agents.judge.trainable = False
|
|
306
309
|
|
|
307
|
-
async def run(self, task: vf.Task, agents: vf.Agents) -> None:
|
|
308
|
-
if self.config.share_runtime:
|
|
309
|
-
async with agents.solver.provision(task) as box:
|
|
310
|
-
solution = await agents.solver.run(task, runtime=box)
|
|
311
|
-
judge_task = JudgeTask.from_trace(solution, self.config.task)
|
|
312
|
-
await agents.judge.run(judge_task, runtime=box)
|
|
313
|
-
return
|
|
314
|
-
|
|
315
|
-
solution = await agents.solver.run(task)
|
|
316
|
-
if not solution.ok:
|
|
317
|
-
return
|
|
318
|
-
await agents.judge.run(
|
|
319
|
-
JudgeTask.from_trace(solution, self.config.task, share_runtime=False)
|
|
320
|
-
)
|
|
321
|
-
|
|
322
310
|
async def finalize(self, task: vf.Task, episode: vf.Episode) -> None:
|
|
323
311
|
by_agent = {t.agent.name: t for t in episode.traces}
|
|
324
312
|
if "judge" not in by_agent:
|
|
@@ -336,3 +324,25 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
|
|
|
336
324
|
total = sum(criterion.weight for criterion in criteria)
|
|
337
325
|
reward = sum(c.weight * scores[c.name] for c in criteria) / total
|
|
338
326
|
solution.record_reward("judge", reward, weight=self.config.score.judge_weight)
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
class SharedAgenticJudgeEnv(AgenticJudgeEnv):
|
|
330
|
+
"""Judge the solver in its runtime, preserving the complete mutable workspace."""
|
|
331
|
+
|
|
332
|
+
async def run(self, task: vf.Task, agents: vf.Agents) -> None:
|
|
333
|
+
async with agents.solver.provision(task) as box:
|
|
334
|
+
solution = await agents.solver.run(task, runtime=box)
|
|
335
|
+
judge_task = JudgeTask.from_trace(solution, self.config.task)
|
|
336
|
+
await agents.judge.run(judge_task, runtime=box)
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
class IsolatedAgenticJudgeEnv(AgenticJudgeEnv):
|
|
340
|
+
"""Judge only collected artifacts in a fresh box with the solver's policy."""
|
|
341
|
+
|
|
342
|
+
async def run(self, task: vf.Task, agents: vf.Agents) -> None:
|
|
343
|
+
solution = await agents.solver.run(task)
|
|
344
|
+
if not solution.ok:
|
|
345
|
+
return
|
|
346
|
+
await agents.judge.run(
|
|
347
|
+
JudgeTask.from_trace(solution, self.config.task, share_runtime=False)
|
|
348
|
+
)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.dev87
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -211,7 +211,7 @@ verifiers/v1/configs/__init__.py,sha256=X7u6X7B3ieD1HZl79Tv1kg-yPHbT0QLvAebaGqxV
|
|
|
211
211
|
verifiers/v1/configs/agent.py,sha256=V_2FDvPHRB3_bHN9QnNnDYOjlmTP7NtG22SaopU-X64,3261
|
|
212
212
|
verifiers/v1/configs/client.py,sha256=dCi2aC_wVHCrM0xRi9foFcbqp9pN2GR7yxEs1z_ChJM,4371
|
|
213
213
|
verifiers/v1/configs/env.py,sha256=7ESsWZRWJpKqR5EEV1rmj0R1Dkn2Pn8MVSXPJOzRiF8,7885
|
|
214
|
-
verifiers/v1/configs/harness.py,sha256=
|
|
214
|
+
verifiers/v1/configs/harness.py,sha256=GuXpAEPTwP2i2hp0bM4CYzC1yAMz-cp_j2_4-Qp5eM4,1714
|
|
215
215
|
verifiers/v1/configs/judge.py,sha256=jc9w8eYsqSEyMMY7n12EWyYcwqQl5s-47xV3Me7WA0c,2242
|
|
216
216
|
verifiers/v1/configs/legacy.py,sha256=YOJM6L5u6-jYpZ1CLpPInufKdmz1lEsMNKIRdT2JaaY,1080
|
|
217
217
|
verifiers/v1/configs/retries.py,sha256=-nPlgmz7J_NZ4GOYG6oCc3thXo4La2SufBmvQGvvPGo,821
|
|
@@ -231,10 +231,11 @@ verifiers/v1/dialects/base.py,sha256=YZQDnasQseDXK-GHLfJLZgyHIIygS80YhD_sTjmUbfA
|
|
|
231
231
|
verifiers/v1/dialects/chat.py,sha256=R_otqi5N0snB1QVKIYloNeLHR9kc7OmI06HYg1ir_Lw,14021
|
|
232
232
|
verifiers/v1/dialects/responses.py,sha256=ismnxUikPcPUIm80FUTPBbj-Cn1e0yGAsISMEseAWRc,13679
|
|
233
233
|
verifiers/v1/envs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
234
|
-
verifiers/v1/envs/agentic_judge/__init__.py,sha256=
|
|
235
|
-
verifiers/v1/envs/agentic_judge/env.py,sha256=
|
|
234
|
+
verifiers/v1/envs/agentic_judge/__init__.py,sha256=LfEcRHLG_siKzhzWmODDXWSLpLgfum3UFdQDK9XG0gk,325
|
|
235
|
+
verifiers/v1/envs/agentic_judge/env.py,sha256=9HdI8dm6m5QTNcHhpMvu6JJ2GlqH6e6xrc0inrEEZMk,14443
|
|
236
236
|
verifiers/v1/envs/best_of_n/__init__.py,sha256=mubASiwXMNIhnQFDccNAc7yZ0OFsOVK8hH8WIx9b2A4,119
|
|
237
237
|
verifiers/v1/envs/best_of_n/env.py,sha256=KEUJBCVqblnXP_lEwm3mjGdaikBMkecz8UjOFPBhxc0,1987
|
|
238
|
+
verifiers/v1/envs/shared_agentic_judge/__init__.py,sha256=9T4vyYAuqgZSQjqTIceghxnpauvqeW6XGhZSQDCttMA,168
|
|
238
239
|
verifiers/v1/envs/single_agent/__init__.py,sha256=r0edAc_6gtHBhPzeBGVcdHZZqSdCPZnLMje6eVgyz5U,138
|
|
239
240
|
verifiers/v1/envs/single_agent/env.py,sha256=lPs_a6VA7KoH1xqA3T8S32Jzb_d4zrx0_54LrqT5z1Q,1187
|
|
240
241
|
verifiers/v1/envs/user_sim/__init__.py,sha256=nnT695HdqjAEouYL5wtVu6MhIp2wf5yuK-EYgxOVx0s,118
|
|
@@ -337,8 +338,8 @@ verifiers/v1/utils/platform.py,sha256=56Ixmk1SER6q5LvzyYcA0hgzGpzhxKZL8Hiqp4c_XV
|
|
|
337
338
|
verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
|
|
338
339
|
verifiers/v1/utils/score.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
|
|
339
340
|
verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
|
|
340
|
-
verifiers-0.2.2.
|
|
341
|
-
verifiers-0.2.2.
|
|
342
|
-
verifiers-0.2.2.
|
|
343
|
-
verifiers-0.2.2.
|
|
344
|
-
verifiers-0.2.2.
|
|
341
|
+
verifiers-0.2.2.dev87.dist-info/METADATA,sha256=U5lglk-lfizY1P98ZGz92ItcVHUBniDOdYTDLrExpaE,4545
|
|
342
|
+
verifiers-0.2.2.dev87.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
343
|
+
verifiers-0.2.2.dev87.dist-info/entry_points.txt,sha256=v6v0QT9vVExnfn4br42MostI_o3f3ZbWYlTormz-U3g,515
|
|
344
|
+
verifiers-0.2.2.dev87.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
345
|
+
verifiers-0.2.2.dev87.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|