verifiers 0.2.2.dev51__py3-none-any.whl → 0.2.2.dev53__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/__init__.py +11 -0
- verifiers/v1/artifacts.py +144 -0
- verifiers/v1/envs/agentic_judge/__init__.py +2 -0
- verifiers/v1/envs/agentic_judge/env.py +120 -52
- verifiers/v1/state.py +3 -2
- verifiers/v1/task.py +6 -0
- verifiers/v1/tasksets/harbor/taskset.py +99 -0
- verifiers/v1/utils/git.py +74 -9
- {verifiers-0.2.2.dev51.dist-info → verifiers-0.2.2.dev53.dist-info}/METADATA +1 -1
- {verifiers-0.2.2.dev51.dist-info → verifiers-0.2.2.dev53.dist-info}/RECORD +13 -12
- {verifiers-0.2.2.dev51.dist-info → verifiers-0.2.2.dev53.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev51.dist-info → verifiers-0.2.2.dev53.dist-info}/entry_points.txt +0 -0
- {verifiers-0.2.2.dev51.dist-info → verifiers-0.2.2.dev53.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/__init__.py
CHANGED
|
@@ -4,6 +4,12 @@ from pydantic_config import BaseConfig
|
|
|
4
4
|
|
|
5
5
|
from verifiers.v1.acp import ACP
|
|
6
6
|
from verifiers.v1.agent import Agent, Agents, Interaction, Segment, make_agent
|
|
7
|
+
from verifiers.v1.artifacts import (
|
|
8
|
+
ARTIFACTS_DIR,
|
|
9
|
+
Artifact,
|
|
10
|
+
collect,
|
|
11
|
+
restore,
|
|
12
|
+
)
|
|
7
13
|
from verifiers.v1.clients import (
|
|
8
14
|
BaseClientConfig,
|
|
9
15
|
Client,
|
|
@@ -297,6 +303,11 @@ __all__ = [ # noqa: RUF022 - grouped by public API area
|
|
|
297
303
|
"PATCH_CAP_BYTES",
|
|
298
304
|
"capture_patch",
|
|
299
305
|
"resolve_head",
|
|
306
|
+
# grading artifacts
|
|
307
|
+
"ARTIFACTS_DIR",
|
|
308
|
+
"Artifact",
|
|
309
|
+
"collect",
|
|
310
|
+
"restore",
|
|
300
311
|
# scoring
|
|
301
312
|
"compare_stdout_results",
|
|
302
313
|
"extract_boxed_answer",
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
"""Artifact collection and restoration across runtimes."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
import shlex
|
|
7
|
+
import uuid
|
|
8
|
+
from pathlib import PurePosixPath
|
|
9
|
+
from typing import TYPE_CHECKING
|
|
10
|
+
|
|
11
|
+
from pydantic import Field
|
|
12
|
+
|
|
13
|
+
from verifiers.v1.types import StrictBaseModel
|
|
14
|
+
|
|
15
|
+
if TYPE_CHECKING:
|
|
16
|
+
from verifiers.v1.runtimes import Runtime
|
|
17
|
+
|
|
18
|
+
logger = logging.getLogger(__name__)
|
|
19
|
+
|
|
20
|
+
ARTIFACTS_DIR = "/logs/artifacts"
|
|
21
|
+
"""Implicit artifact directory; tasks that write here need no declaration."""
|
|
22
|
+
|
|
23
|
+
MAX_ARTIFACT_BYTES = 32 * 1024 * 1024
|
|
24
|
+
"""Ceiling per collection. Sized for a delta, not a tree: the grading box boots from the
|
|
25
|
+
agent's image, so the repo is already there and only its output has to travel."""
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class Artifact(StrictBaseModel):
|
|
29
|
+
"""One path to restore at the same location in another runtime."""
|
|
30
|
+
|
|
31
|
+
source: str
|
|
32
|
+
exclude: list[str] = Field(default_factory=list)
|
|
33
|
+
"""`tar --exclude` patterns, applied when `source` is a directory."""
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
async def collect(
|
|
37
|
+
runtime: Runtime, artifacts: list[Artifact] | None = None
|
|
38
|
+
) -> dict[str, bytes]:
|
|
39
|
+
"""Tar the convention dir and every declared path out of `runtime`.
|
|
40
|
+
|
|
41
|
+
Keyed by source path; the values are tar archives. Insertion order is the order
|
|
42
|
+
they were declared, and a path cannot be collected twice.
|
|
43
|
+
|
|
44
|
+
A declared source that is missing raises: it was declared because grading needs it,
|
|
45
|
+
and grading a partial state scores the rollout wrong rather than failing it. The
|
|
46
|
+
implicit convention sweep is exempt — most tasks never write there.
|
|
47
|
+
|
|
48
|
+
Each source is archived separately so its exclude patterns stay local.
|
|
49
|
+
"""
|
|
50
|
+
# Resolve relative sources against the runtime workdir. Joining also normalises
|
|
51
|
+
# `/work/` to `/work`, so one tree cannot key two entries (the source is both the
|
|
52
|
+
# dict key and `restore`'s rm -rf target).
|
|
53
|
+
workdir = PurePosixPath(getattr(runtime.config, "workdir", "") or "/")
|
|
54
|
+
declared = [
|
|
55
|
+
a.model_copy(update={"source": str(workdir / a.source)})
|
|
56
|
+
for a in artifacts or []
|
|
57
|
+
]
|
|
58
|
+
convention = PurePosixPath(ARTIFACTS_DIR)
|
|
59
|
+
sweep = not any(
|
|
60
|
+
(p := PurePosixPath(a.source)) == convention
|
|
61
|
+
or p.is_relative_to(convention)
|
|
62
|
+
or convention.is_relative_to(p)
|
|
63
|
+
for a in declared
|
|
64
|
+
)
|
|
65
|
+
entries = ([Artifact(source=ARTIFACTS_DIR)] if sweep else []) + declared
|
|
66
|
+
|
|
67
|
+
collected: dict[str, bytes] = {}
|
|
68
|
+
budget = MAX_ARTIFACT_BYTES
|
|
69
|
+
for artifact in entries:
|
|
70
|
+
source = artifact.source
|
|
71
|
+
if (await runtime.run(["test", "-e", source], {})).exit_code != 0:
|
|
72
|
+
if sweep and source == ARTIFACTS_DIR:
|
|
73
|
+
continue
|
|
74
|
+
raise RuntimeError(
|
|
75
|
+
f"declared artifact {source!r} does not exist in the runtime"
|
|
76
|
+
)
|
|
77
|
+
archive = await _tar_out(runtime, artifact, budget)
|
|
78
|
+
budget -= len(archive)
|
|
79
|
+
collected[source] = archive
|
|
80
|
+
|
|
81
|
+
logger.debug("collected artifact roots: %s", list(collected))
|
|
82
|
+
return collected
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
async def restore(runtime: Runtime, collected: dict[str, bytes]) -> None:
|
|
86
|
+
"""Extract `collected` in `runtime` at the original absolute paths."""
|
|
87
|
+
if not collected:
|
|
88
|
+
return
|
|
89
|
+
# Restoring into the subprocess runtime would extract absolute paths onto the
|
|
90
|
+
# developer's filesystem, so refuse it before any archive reaches the host.
|
|
91
|
+
if getattr(runtime.config, "type", None) == "subprocess":
|
|
92
|
+
raise RuntimeError(
|
|
93
|
+
"refusing to restore artifacts into the subprocess runtime: extraction "
|
|
94
|
+
"writes to absolute paths on the host. Grade in a container."
|
|
95
|
+
)
|
|
96
|
+
# Clear every root up front, not per entry: a later nested root would otherwise
|
|
97
|
+
# delete content an earlier one just restored. Clearing also drops any file or
|
|
98
|
+
# symlink the image left at the target.
|
|
99
|
+
roots = " ".join(shlex.quote(root) for root in collected)
|
|
100
|
+
await _run(runtime, f"rm -rf -- {roots}", "clear artifact roots")
|
|
101
|
+
for root, archive in collected.items():
|
|
102
|
+
path = f"/tmp/vf-artifact-{uuid.uuid4().hex}.tar"
|
|
103
|
+
await runtime.write(path, archive)
|
|
104
|
+
await _run(
|
|
105
|
+
runtime,
|
|
106
|
+
f"tar -xf {shlex.quote(path)} -C / && rm -f {shlex.quote(path)}",
|
|
107
|
+
f"restore artifact {root!r}",
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
async def _tar_out(runtime: Runtime, artifact: Artifact, budget: int) -> bytes:
|
|
112
|
+
path = f"/tmp/vf-artifact-{uuid.uuid4().hex}.tar"
|
|
113
|
+
excludes = " ".join(f"--exclude={shlex.quote(p)}" for p in artifact.exclude)
|
|
114
|
+
try:
|
|
115
|
+
await _run(
|
|
116
|
+
runtime,
|
|
117
|
+
f"tar -cf {shlex.quote(path)} -C / {excludes} -- "
|
|
118
|
+
f"{shlex.quote(artifact.source.lstrip('/'))}",
|
|
119
|
+
f"collect artifact {artifact.source!r}",
|
|
120
|
+
)
|
|
121
|
+
# Size it in the box: an oversized collection is refused before it reaches host
|
|
122
|
+
# memory, not after.
|
|
123
|
+
sized = await runtime.run(["sh", "-c", f"wc -c < {shlex.quote(path)}"], {})
|
|
124
|
+
if (raw := sized.stdout.strip()).isdigit() and int(raw) > budget:
|
|
125
|
+
raise RuntimeError(
|
|
126
|
+
f"artifact {artifact.source!r} takes the collection over the "
|
|
127
|
+
f"{MAX_ARTIFACT_BYTES} byte limit. The grading box boots from the "
|
|
128
|
+
"agent's image, so only the delta needs to travel — narrow the source "
|
|
129
|
+
"or add `exclude` patterns."
|
|
130
|
+
)
|
|
131
|
+
return await runtime.read(path)
|
|
132
|
+
finally:
|
|
133
|
+
# Best-effort: the box is about to be destroyed and the name is unique per call.
|
|
134
|
+
try:
|
|
135
|
+
await runtime.run(["rm", "-f", path], {})
|
|
136
|
+
except Exception:
|
|
137
|
+
logger.debug("failed to remove %s", path, exc_info=True)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
async def _run(runtime: Runtime, command: str, action: str) -> None:
|
|
141
|
+
result = await runtime.run(["sh", "-c", command], {})
|
|
142
|
+
if result.exit_code:
|
|
143
|
+
detail = (result.stderr or result.stdout).strip()[-500:]
|
|
144
|
+
raise RuntimeError(f"failed to {action}: {detail}")
|
|
@@ -1,14 +1,16 @@
|
|
|
1
|
-
"""agentic-judge: a solver plays the task, a judge verifies
|
|
1
|
+
"""agentic-judge: a solver plays the task, then a judge verifies the work.
|
|
2
2
|
|
|
3
|
-
A reusable env (`--env.id agentic-judge` over any taskset)
|
|
4
|
-
provisioned from
|
|
5
|
-
and a code-executing judge then inspects the work as the agent left it, with
|
|
6
|
-
the solver's full trace record uploaded at `/tmp/trace.json`. The judge grades
|
|
3
|
+
A reusable env (`--env.id agentic-judge` over any taskset). The solver plays the
|
|
4
|
+
task in a container provisioned from its runtime policy; the judge then grades
|
|
7
5
|
rubric criteria (`[env.task]`: policy prompt, criteria file) and writes its
|
|
8
|
-
verdicts to `/tmp/verdict.json
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
default).
|
|
6
|
+
verdicts to `/tmp/verdict.json`, with the solver's full trace record uploaded at
|
|
7
|
+
`/tmp/trace.json`. `finalize()` validates them strictly onto the solver's trace —
|
|
8
|
+
`judge/<name>` metrics plus a weighted-mean `judge` reward, composed with the
|
|
9
|
+
taskset's own rewards via `[env.score]` (judge-only by default).
|
|
10
|
+
|
|
11
|
+
`--env.share-runtime` controls whether the judge uses the solver's runtime. It is
|
|
12
|
+
enabled by default. When disabled, the judge gets a fresh runtime containing the
|
|
13
|
+
task's collected artifacts.
|
|
12
14
|
"""
|
|
13
15
|
|
|
14
16
|
import json
|
|
@@ -21,6 +23,7 @@ from pydantic import Field, field_validator
|
|
|
21
23
|
|
|
22
24
|
import verifiers.v1 as vf
|
|
23
25
|
from verifiers.v1.types import StrictBaseModel
|
|
26
|
+
from verifiers.v1.utils.compile import validate_pairing
|
|
24
27
|
|
|
25
28
|
VERDICT_FILE = "/tmp/verdict.json"
|
|
26
29
|
TRACE_FILE = "/tmp/trace.json"
|
|
@@ -97,19 +100,31 @@ def _render(template: str, **fields: str) -> str:
|
|
|
97
100
|
return pattern.sub(lambda m: fields[m.group(1)], template)
|
|
98
101
|
|
|
99
102
|
|
|
100
|
-
|
|
103
|
+
_RECORD_NOTE = f"""\
|
|
104
|
+
The agent's raw trace record (JSON: messages, tool calls, and its `info`
|
|
105
|
+
artifacts) is written by the harness — not the agent — at `{TRACE_FILE}`. The
|
|
106
|
+
record can be very large — never dump it whole; peek selectively (list its
|
|
107
|
+
keys, then slice out specific fields with python or jq) and pull only what you
|
|
108
|
+
need. It is complete — it may also carry the task's own scores/metrics and
|
|
109
|
+
reference material (a gold answer, a reference solution, held-out tests). Those
|
|
110
|
+
are context, not your standard: recorded scores can be wrong and references can
|
|
111
|
+
be narrower than the task; do not over-index on how a reference solves it. Your
|
|
112
|
+
verdict is what YOU verified by execution."""
|
|
113
|
+
|
|
114
|
+
SHARED_WORKSPACE_NOTE = f"""\
|
|
115
|
+
## Your workspace
|
|
116
|
+
|
|
117
|
+
The graded agent worked in this sandbox. Its edits and any scoring side effects
|
|
118
|
+
are present. {_RECORD_NOTE}"""
|
|
119
|
+
|
|
120
|
+
ISOLATED_WORKSPACE_NOTE = f"""\
|
|
101
121
|
## Your workspace
|
|
102
122
|
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
and pull only what you need. It is complete — it may also carry the task's own
|
|
109
|
-
scores/metrics and reference material (a gold answer, a reference solution,
|
|
110
|
-
held-out tests). Those are context, not your standard: recorded scores can be
|
|
111
|
-
wrong and references can be narrower than the task; do not over-index on how a
|
|
112
|
-
reference solves it. Your verdict is what YOU verified by execution."""
|
|
123
|
+
This is a fresh sandbox built from the task's image. The task's published
|
|
124
|
+
artifacts were restored at their original paths; other changes made by the
|
|
125
|
+
graded agent are not present. The usual artifact location is
|
|
126
|
+
`{vf.ARTIFACTS_DIR}/` (for code tasks, typically a patch to read or apply), plus
|
|
127
|
+
any task-declared paths. {_RECORD_NOTE}"""
|
|
113
128
|
|
|
114
129
|
HINT_SECTION = """\
|
|
115
130
|
## Hints
|
|
@@ -125,13 +140,30 @@ class JudgeTask(vf.Task):
|
|
|
125
140
|
|
|
126
141
|
NEEDS_CONTAINER = True
|
|
127
142
|
|
|
128
|
-
def __init__(
|
|
143
|
+
def __init__(
|
|
144
|
+
self,
|
|
145
|
+
data: vf.TaskData,
|
|
146
|
+
files: dict[str, bytes],
|
|
147
|
+
artifacts: dict[str, bytes],
|
|
148
|
+
) -> None:
|
|
129
149
|
super().__init__(data)
|
|
130
150
|
self.files = files
|
|
151
|
+
self.artifacts = artifacts
|
|
131
152
|
|
|
132
153
|
@classmethod
|
|
133
|
-
def from_trace(
|
|
134
|
-
|
|
154
|
+
def from_trace(
|
|
155
|
+
cls,
|
|
156
|
+
solution: vf.Trace,
|
|
157
|
+
config: "JudgeTaskConfig",
|
|
158
|
+
share_runtime: bool = True,
|
|
159
|
+
) -> "JudgeTask":
|
|
160
|
+
"""Mint the judge's task from the solver's finished trace.
|
|
161
|
+
|
|
162
|
+
`share_runtime` selects both the workspace note and artifact transport. In
|
|
163
|
+
the solver's box the published artifacts are already on disk, so none
|
|
164
|
+
travel; a fresh box gets the collected set, restored by `setup` at the
|
|
165
|
+
paths they had.
|
|
166
|
+
"""
|
|
135
167
|
solved = solution.task.data
|
|
136
168
|
files = {TRACE_FILE: json.dumps(solution.to_record()).encode()}
|
|
137
169
|
template = config.build_prompt()
|
|
@@ -139,7 +171,10 @@ class JudgeTask(vf.Task):
|
|
|
139
171
|
if "{prompt}" not in template:
|
|
140
172
|
# A policy that doesn't place the task statement itself still needs it.
|
|
141
173
|
body += "\n\n" + _render(TASK_SECTION, prompt=solved.prompt_text)
|
|
142
|
-
|
|
174
|
+
workspace_note = (
|
|
175
|
+
SHARED_WORKSPACE_NOTE if share_runtime else ISOLATED_WORKSPACE_NOTE
|
|
176
|
+
)
|
|
177
|
+
sections = [body, _verdict_section(config.criteria()), workspace_note]
|
|
143
178
|
if (hint := config.build_hint()) is not None:
|
|
144
179
|
sections.insert(1, _render(HINT_SECTION, hint=hint))
|
|
145
180
|
prompt = "\n\n".join(sections)
|
|
@@ -152,9 +187,11 @@ class JudgeTask(vf.Task):
|
|
|
152
187
|
resources=solved.resources,
|
|
153
188
|
),
|
|
154
189
|
files=files,
|
|
190
|
+
artifacts={} if share_runtime else solution.state.artifacts,
|
|
155
191
|
)
|
|
156
192
|
|
|
157
193
|
async def setup(self, trace: vf.Trace, runtime: vf.Runtime) -> None:
|
|
194
|
+
await vf.restore(runtime, self.artifacts)
|
|
158
195
|
# The solver had this box first: a pre-seeded verdict must never read as
|
|
159
196
|
# the judge's own, and a file (or planted symlink) at an upload path must
|
|
160
197
|
# never survive it — a symlinked TRACE_FILE would redirect the write onto
|
|
@@ -177,34 +214,53 @@ class JudgeTask(vf.Task):
|
|
|
177
214
|
trace.info["verdict"] = json.loads(raw)
|
|
178
215
|
|
|
179
216
|
|
|
217
|
+
class TextFile(vf.BaseConfig):
|
|
218
|
+
"""An explicit file-backed text value for config formats without `Path` values."""
|
|
219
|
+
|
|
220
|
+
path: Path
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
TextSource = str | Path | TextFile
|
|
224
|
+
|
|
225
|
+
|
|
180
226
|
class JudgeTaskConfig(vf.BaseConfig):
|
|
181
227
|
"""The judge's minted task: the grading policy and what lands in its box."""
|
|
182
228
|
|
|
183
|
-
prompt:
|
|
184
|
-
"""Grading
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
"""
|
|
189
|
-
|
|
190
|
-
|
|
229
|
+
prompt: TextSource | None = None
|
|
230
|
+
"""Grading policy: a string is inline text; a `Path` in Python or
|
|
231
|
+
`{ path = "policy.md" }` in config reads a file. Replaces only the policy
|
|
232
|
+
body — the verdict contract and workspace note are always appended. May
|
|
233
|
+
reference `{prompt}` (the solver task's prompt); if it doesn't, the task
|
|
234
|
+
statement is appended after."""
|
|
235
|
+
hint: TextSource | None = None
|
|
236
|
+
"""Optional hints, with the same explicit inline/file forms as `prompt`,
|
|
237
|
+
injected as their own section: task-family pointers into the trace or box —
|
|
238
|
+
e.g. for math, where the reference answer lives in the record; for SWE, to
|
|
239
|
+
diff the repo or read `info.patch`."""
|
|
191
240
|
rubric: Path | None = None
|
|
192
241
|
"""Criteria the judge grades against: a `.toml`/`.json` file with a
|
|
193
242
|
`criteria` list — the plugged rubric judge's format, so the same rubric
|
|
194
243
|
files work for both. None grades the single built-in `solved` criterion."""
|
|
195
244
|
|
|
245
|
+
@staticmethod
|
|
246
|
+
def _resolve(value: TextSource) -> str:
|
|
247
|
+
if isinstance(value, str):
|
|
248
|
+
return value
|
|
249
|
+
path = value if isinstance(value, Path) else value.path
|
|
250
|
+
return path.read_text(encoding="utf-8")
|
|
251
|
+
|
|
196
252
|
def build_prompt(self) -> str:
|
|
197
253
|
if self.prompt is None:
|
|
198
254
|
return GRADE_PROMPT + "\n\n" + TASK_SECTION
|
|
199
|
-
return self.prompt
|
|
255
|
+
return self._resolve(self.prompt)
|
|
200
256
|
|
|
201
257
|
def build_hint(self) -> str | None:
|
|
202
|
-
return self.hint
|
|
258
|
+
return self._resolve(self.hint) if self.hint is not None else None
|
|
203
259
|
|
|
204
260
|
def criteria(self) -> list[Criterion]:
|
|
205
261
|
if self.rubric is None:
|
|
206
262
|
return [SOLVED]
|
|
207
|
-
text = self.rubric.read_text()
|
|
263
|
+
text = self.rubric.read_text(encoding="utf-8")
|
|
208
264
|
data = (
|
|
209
265
|
tomllib.loads(text)
|
|
210
266
|
if self.rubric.suffix.lower() == ".toml"
|
|
@@ -241,23 +297,23 @@ class ScoreConfig(vf.BaseConfig):
|
|
|
241
297
|
|
|
242
298
|
class AgenticJudgeEnvConfig(vf.EnvConfig):
|
|
243
299
|
solver: vf.AgentConfig = vf.AgentConfig()
|
|
244
|
-
"""The solver agent.
|
|
245
|
-
|
|
300
|
+
"""The solver agent. Its runtime must be a container:
|
|
301
|
+
`--env.solver.runtime.type docker|prime`."""
|
|
246
302
|
judge: vf.AgentConfig = vf.AgentConfig()
|
|
247
|
-
"""The judge agent.
|
|
248
|
-
|
|
303
|
+
"""The judge agent. Its runtime is ignored when `share_runtime` is enabled;
|
|
304
|
+
otherwise it must be a container."""
|
|
305
|
+
share_runtime: bool = True
|
|
306
|
+
"""Whether the judge grades in the solver's runtime."""
|
|
249
307
|
task: JudgeTaskConfig = JudgeTaskConfig()
|
|
250
308
|
score: ScoreConfig = ScoreConfig()
|
|
251
309
|
|
|
252
310
|
|
|
253
311
|
class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
|
|
254
312
|
def __init__(self, config: AgenticJudgeEnvConfig) -> None:
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
update={"runtime": config.solver.runtime}
|
|
260
|
-
)
|
|
313
|
+
if config.share_runtime:
|
|
314
|
+
config.judge = config.judge.model_copy(
|
|
315
|
+
update={"runtime": config.solver.runtime}
|
|
316
|
+
)
|
|
261
317
|
super().__init__(config)
|
|
262
318
|
self._check_agents()
|
|
263
319
|
# A missing policy file or a malformed rubric fails here, not mid-episode.
|
|
@@ -271,30 +327,42 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
|
|
|
271
327
|
judge = self._harnesses["judge"]
|
|
272
328
|
if not judge.EXECUTES_CODE:
|
|
273
329
|
raise ValueError(
|
|
274
|
-
"agentic-judge
|
|
330
|
+
"agentic-judge requires a judge harness that can execute code, but "
|
|
275
331
|
f"harness {judge.config.id!r} is a tool-less chat loop — a verdict "
|
|
276
332
|
"that needs no execution is a plugged judge "
|
|
277
333
|
"(--env.taskset.task.judges), not an agent."
|
|
278
334
|
)
|
|
279
335
|
if isinstance(self.config.solver.runtime, vf.SubprocessConfig):
|
|
280
336
|
raise TypeError(
|
|
281
|
-
"agentic-judge
|
|
282
|
-
"
|
|
337
|
+
"agentic-judge requires the solver to run in a container, but it "
|
|
338
|
+
"resolves to the subprocess runtime; use "
|
|
283
339
|
"--env.solver.runtime.type docker or prime"
|
|
284
340
|
)
|
|
341
|
+
validate_pairing(judge, JudgeTask, self.config.judge.runtime)
|
|
285
342
|
|
|
286
343
|
async def setup(self, agents: vf.Agents) -> None:
|
|
287
344
|
# The judge grades the policy; its tokens are never training data.
|
|
288
345
|
agents.judge.trainable = False
|
|
289
346
|
|
|
290
347
|
async def run(self, task: vf.Task, agents: vf.Agents) -> None:
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
348
|
+
if self.config.share_runtime:
|
|
349
|
+
async with agents.solver.provision(task) as box:
|
|
350
|
+
solution = await agents.solver.run(task, runtime=box)
|
|
351
|
+
judge_task = JudgeTask.from_trace(solution, self.config.task)
|
|
352
|
+
await agents.judge.run(judge_task, runtime=box)
|
|
353
|
+
return
|
|
354
|
+
|
|
355
|
+
solution = await agents.solver.run(task)
|
|
356
|
+
if not solution.ok:
|
|
357
|
+
return
|
|
358
|
+
await agents.judge.run(
|
|
359
|
+
JudgeTask.from_trace(solution, self.config.task, share_runtime=False)
|
|
360
|
+
)
|
|
295
361
|
|
|
296
362
|
async def finalize(self, task: vf.Task, episode: vf.Episode) -> None:
|
|
297
363
|
by_agent = {t.agent.name: t for t in episode.traces}
|
|
364
|
+
if "judge" not in by_agent:
|
|
365
|
+
return
|
|
298
366
|
solution, verdict = by_agent["solver"], by_agent["judge"]
|
|
299
367
|
data = verdict.info.get("verdict")
|
|
300
368
|
if not isinstance(data, dict) or not isinstance(data.get("verdicts"), list):
|
verifiers/v1/state.py
CHANGED
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
"""Mutable state shared within one rollout.
|
|
2
2
|
|
|
3
3
|
Tool servers synchronize it through the interception state channel. It is excluded
|
|
4
|
-
from serialized traces
|
|
4
|
+
from serialized traces.
|
|
5
5
|
"""
|
|
6
6
|
|
|
7
|
-
from pydantic import ConfigDict
|
|
7
|
+
from pydantic import ConfigDict, Field
|
|
8
8
|
from typing_extensions import TypeVar
|
|
9
9
|
|
|
10
10
|
from verifiers.v1.types import StrictBaseModel
|
|
@@ -13,6 +13,7 @@ from verifiers.v1.utils.generic import concrete_type
|
|
|
13
13
|
|
|
14
14
|
class State(StrictBaseModel):
|
|
15
15
|
model_config = ConfigDict(ser_json_inf_nan="constants")
|
|
16
|
+
artifacts: dict[str, bytes] = Field(default_factory=dict)
|
|
16
17
|
|
|
17
18
|
|
|
18
19
|
StateT = TypeVar("StateT", bound=State, default=State)
|
verifiers/v1/task.py
CHANGED
|
@@ -34,6 +34,7 @@ from pydantic import ConfigDict, Field
|
|
|
34
34
|
from pydantic_config import BaseConfig
|
|
35
35
|
from typing_extensions import TypeVar
|
|
36
36
|
|
|
37
|
+
from verifiers.v1.artifacts import Artifact
|
|
37
38
|
from verifiers.v1.configs.task import TaskConfig
|
|
38
39
|
from verifiers.v1.decorators import discover_decorated, invoke_all
|
|
39
40
|
from verifiers.v1.errors import TaskError, boundary
|
|
@@ -102,6 +103,11 @@ class TaskData(StrictBaseModel):
|
|
|
102
103
|
"""Execution-time destinations denied by this task and combined with runtime
|
|
103
104
|
blocks. Non-empty concrete allowlists cannot be combined with blocklists. Docker
|
|
104
105
|
framework routes take precedence; ordinary Prime deny rules pass through unchanged."""
|
|
106
|
+
artifacts: list[Artifact] = Field(default_factory=list)
|
|
107
|
+
"""Paths collected from one runtime and restored at the same locations in another,
|
|
108
|
+
on top of the implicitly collected `/logs/artifacts/` convention dir. Declare
|
|
109
|
+
runtime outputs that must cross that boundary. A declared path that is missing at
|
|
110
|
+
collection time fails the rollout."""
|
|
105
111
|
timeout: TaskTimeout = TaskTimeout()
|
|
106
112
|
resources: TaskResources = TaskResources()
|
|
107
113
|
|
|
@@ -10,6 +10,7 @@ deliberately uses the harness runtime image. Tasks without an environment also u
|
|
|
10
10
|
image unless ``require_image`` is set.
|
|
11
11
|
"""
|
|
12
12
|
|
|
13
|
+
import asyncio
|
|
13
14
|
import hashlib
|
|
14
15
|
import io
|
|
15
16
|
import shutil
|
|
@@ -23,12 +24,14 @@ from pathlib import Path
|
|
|
23
24
|
|
|
24
25
|
from pydantic import Field
|
|
25
26
|
|
|
27
|
+
from verifiers.v1.artifacts import Artifact, collect
|
|
26
28
|
from verifiers.v1.configs.taskset import TasksetConfig
|
|
27
29
|
from verifiers.v1.decorators import reward
|
|
28
30
|
from verifiers.v1.errors import SandboxError
|
|
29
31
|
from verifiers.v1.runtimes import Runtime
|
|
30
32
|
from verifiers.v1.task import Task, TaskData, TaskResources, TaskTimeout
|
|
31
33
|
from verifiers.v1.taskset import Taskset
|
|
34
|
+
from verifiers.v1.trace import Trace
|
|
32
35
|
from verifiers.v1.types import StrictBaseModel
|
|
33
36
|
|
|
34
37
|
CACHE = Path.home() / ".cache" / "harbor"
|
|
@@ -75,6 +78,13 @@ class Author(StrictBaseModel):
|
|
|
75
78
|
email: str | None = None
|
|
76
79
|
|
|
77
80
|
|
|
81
|
+
class CollectHook(StrictBaseModel):
|
|
82
|
+
"""One `[[verifier.collect]]` command, run in the agent's box by `finalize`."""
|
|
83
|
+
|
|
84
|
+
command: str
|
|
85
|
+
timeout_sec: float = 600.0
|
|
86
|
+
|
|
87
|
+
|
|
78
88
|
class HarborData(TaskData):
|
|
79
89
|
"""Parsed ``task.toml`` metadata plus the host-side verifier directory.
|
|
80
90
|
|
|
@@ -93,11 +103,43 @@ class HarborData(TaskData):
|
|
|
93
103
|
"""Raw [verifier.env] entries (literals or `${VAR}`/`${VAR:-default}` templates).
|
|
94
104
|
Resolved against the host environment at scoring time, like `harbor run` — so a
|
|
95
105
|
verifier that needs judge API keys or configuration actually receives them."""
|
|
106
|
+
collect: list[CollectHook] = Field(default_factory=list)
|
|
107
|
+
"""`[[verifier.collect]]` blocks: commands that snapshot runtime state into files
|
|
108
|
+
after the agent stops, so the files can travel to a grading box as artifacts."""
|
|
96
109
|
|
|
97
110
|
|
|
98
111
|
class HarborTask(Task[HarborData]):
|
|
99
112
|
"""Stage and run Harbor's verifier inside the task's live runtime."""
|
|
100
113
|
|
|
114
|
+
async def finalize(self, trace: Trace, runtime: Runtime) -> None:
|
|
115
|
+
"""Run Harbor's collect hooks while the agent's box is still alive.
|
|
116
|
+
|
|
117
|
+
Harbor runs these after the agent phase and before artifact collection, which
|
|
118
|
+
is exactly what `finalize` means here, so the hook maps onto the existing
|
|
119
|
+
lifecycle rather than needing a stage of its own.
|
|
120
|
+
|
|
121
|
+
Strict, unlike `harbor run`, which logs a failed hook and carries on: there the
|
|
122
|
+
output is observability, here it is a grading input, and a silently absent file
|
|
123
|
+
makes the verifier score a stale state instead of failing loudly.
|
|
124
|
+
"""
|
|
125
|
+
for hook in self.data.collect:
|
|
126
|
+
try:
|
|
127
|
+
result = await asyncio.wait_for(
|
|
128
|
+
runtime.run(["sh", "-c", hook.command], {}),
|
|
129
|
+
hook.timeout_sec,
|
|
130
|
+
)
|
|
131
|
+
except TimeoutError as exc:
|
|
132
|
+
raise RuntimeError(
|
|
133
|
+
f"collect hook timed out after {hook.timeout_sec}s: {hook.command}"
|
|
134
|
+
) from exc
|
|
135
|
+
if result.exit_code:
|
|
136
|
+
detail = (result.stderr or result.stdout).strip()[-500:]
|
|
137
|
+
raise RuntimeError(
|
|
138
|
+
f"collect hook failed (exit {result.exit_code}): "
|
|
139
|
+
f"{hook.command}\n{detail}"
|
|
140
|
+
)
|
|
141
|
+
trace.state.artifacts = await collect(runtime, self.data.artifacts)
|
|
142
|
+
|
|
101
143
|
@reward(weight=1.0)
|
|
102
144
|
async def solved(self, runtime: Runtime) -> float:
|
|
103
145
|
await runtime.write(
|
|
@@ -248,6 +290,7 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
|
|
|
248
290
|
|
|
249
291
|
harbor_task = HarborModelTask(task_dir)
|
|
250
292
|
parsed = harbor_task.config
|
|
293
|
+
artifacts, collect = parse_verifier_extras(task_dir, parsed)
|
|
251
294
|
environment = parsed.environment
|
|
252
295
|
network = parsed.agent.explicit_phase_policy() or environment.resolve_baseline()
|
|
253
296
|
task, meta = parsed.task, parsed.metadata
|
|
@@ -316,9 +359,65 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
|
|
|
316
359
|
tags=meta.get("tags", []),
|
|
317
360
|
task_dir=str(task_dir),
|
|
318
361
|
verifier_env=parsed.verifier.env,
|
|
362
|
+
artifacts=artifacts,
|
|
363
|
+
collect=collect,
|
|
319
364
|
)
|
|
320
365
|
|
|
321
366
|
|
|
367
|
+
def parse_verifier_extras(
|
|
368
|
+
task_dir: Path, parsed
|
|
369
|
+
) -> tuple[list[Artifact], list[CollectHook]]:
|
|
370
|
+
"""Parse supported artifact and collect-hook settings."""
|
|
371
|
+
from harbor.constants import MAIN_SERVICE_NAME
|
|
372
|
+
from harbor.models.task.artifacts import (
|
|
373
|
+
effective_artifact_service,
|
|
374
|
+
normalize_artifact_entries,
|
|
375
|
+
)
|
|
376
|
+
|
|
377
|
+
verifier = parsed.verifier
|
|
378
|
+
if verifier.environment is not None:
|
|
379
|
+
raise ValueError(
|
|
380
|
+
f"{task_dir.name}: [verifier.environment] declares a separate verifier "
|
|
381
|
+
"image. Grading runs in a fresh box built from the task's own image, so "
|
|
382
|
+
"only the agent's delta has to travel; a different verifier image needs "
|
|
383
|
+
"the full working tree copied over and isn't supported yet."
|
|
384
|
+
)
|
|
385
|
+
if verifier.user is not None:
|
|
386
|
+
raise ValueError(f"{task_dir.name}: [verifier].user is not supported")
|
|
387
|
+
|
|
388
|
+
artifacts: list[Artifact] = []
|
|
389
|
+
for entry in normalize_artifact_entries(parsed.artifacts):
|
|
390
|
+
if effective_artifact_service(entry) != MAIN_SERVICE_NAME:
|
|
391
|
+
raise ValueError(
|
|
392
|
+
f"{task_dir.name}: artifact {entry.source!r} targets additional "
|
|
393
|
+
f"service {entry.service!r}; verifiers currently supports artifacts "
|
|
394
|
+
"from the main service only"
|
|
395
|
+
)
|
|
396
|
+
# `destination` positions a file in Harbor's host trial directory. Verifiers has
|
|
397
|
+
# no such directory (the trace is the record) and Harbor never lets destination
|
|
398
|
+
# affect verifier-side placement, so it cannot change any grading outcome.
|
|
399
|
+
artifacts.append(
|
|
400
|
+
Artifact(source=entry.source, exclude=list(entry.exclude or []))
|
|
401
|
+
)
|
|
402
|
+
|
|
403
|
+
hooks: list[CollectHook] = []
|
|
404
|
+
for hook in verifier.collect:
|
|
405
|
+
if hook.service != MAIN_SERVICE_NAME:
|
|
406
|
+
raise ValueError(
|
|
407
|
+
f"{task_dir.name}: collect hook targets additional service "
|
|
408
|
+
f"{hook.service!r}; verifiers currently supports collect hooks for "
|
|
409
|
+
"the main service only"
|
|
410
|
+
)
|
|
411
|
+
if hook.user is not None:
|
|
412
|
+
raise ValueError(
|
|
413
|
+
f"{task_dir.name}: collect hook `user` is not supported "
|
|
414
|
+
"(commands run as the runtime's default user)"
|
|
415
|
+
)
|
|
416
|
+
hooks.append(CollectHook(command=hook.command, timeout_sec=hook.timeout_sec))
|
|
417
|
+
|
|
418
|
+
return artifacts, hooks
|
|
419
|
+
|
|
420
|
+
|
|
322
421
|
def verifier_env(task: HarborData) -> dict[str, str]:
|
|
323
422
|
"""Resolve templates at scoring time so host secrets are never serialized."""
|
|
324
423
|
if not task.verifier_env:
|
verifiers/v1/utils/git.py
CHANGED
|
@@ -3,7 +3,8 @@
|
|
|
3
3
|
SWE-style tasksets call `capture_patch` from `Task.finalize` — after the harness
|
|
4
4
|
finishes, while the runtime is live, before scoring mutates the repo (restoring
|
|
5
5
|
test files, switching commits) — so the diff is exactly what the agent produced,
|
|
6
|
-
including edits to test files (intentional: they reveal reward hacking)
|
|
6
|
+
including edits to test files (intentional: they reveal reward hacking), and
|
|
7
|
+
excluding whatever `snapshot_untracked` recorded before the agent started.
|
|
7
8
|
|
|
8
9
|
The diff is taken against `base_commit` when the caller has one — a dataset row
|
|
9
10
|
field, or a SHA recorded with `resolve_head` at setup time and kept in host
|
|
@@ -15,8 +16,11 @@ agent commits.
|
|
|
15
16
|
from __future__ import annotations
|
|
16
17
|
|
|
17
18
|
import uuid
|
|
19
|
+
from pathlib import PurePosixPath
|
|
18
20
|
from typing import TYPE_CHECKING
|
|
19
21
|
|
|
22
|
+
from verifiers.v1.errors import SandboxError
|
|
23
|
+
|
|
20
24
|
if TYPE_CHECKING:
|
|
21
25
|
from verifiers.v1.runtimes import Runtime
|
|
22
26
|
from verifiers.v1.trace import Trace
|
|
@@ -41,10 +45,15 @@ _CAPPED = "/tmp/vf_agent_patch"
|
|
|
41
45
|
# tree staged, so reporting success would hide a state later scoring may trip
|
|
42
46
|
# on; a failed head leaves an empty {capped} (the redirect truncates it before
|
|
43
47
|
# head runs), which would read back as a silently empty patch.
|
|
48
|
+
#
|
|
49
|
+
# The unstage step is deliberately outside that accounting: if it fails the patch is
|
|
50
|
+
# merely as wide as it used to be, which is a worse patch, not a broken rollout. The
|
|
51
|
+
# `$#` test is load-bearing — `git reset -q --` with no pathspec unstages everything.
|
|
44
52
|
_DIFF = (
|
|
45
53
|
"rm -f {full} {capped}; "
|
|
46
54
|
"git add -A; "
|
|
47
55
|
"add_rc=$?; "
|
|
56
|
+
'[ "$#" -gt 0 ] && git reset -q -- "$@"; '
|
|
48
57
|
'git -c core.quotepath=off diff --cached --binary "$VF_DIFF_BASE" > {full}; '
|
|
49
58
|
"diff_rc=$?; "
|
|
50
59
|
"git reset -q; "
|
|
@@ -74,32 +83,84 @@ async def resolve_head(runtime: Runtime, env: dict | None = None) -> str:
|
|
|
74
83
|
return (result.stdout or "").strip()
|
|
75
84
|
|
|
76
85
|
|
|
86
|
+
async def snapshot_untracked(runtime: Runtime, env: dict | None = None) -> list[str]:
|
|
87
|
+
"""The repo's untracked files, to hand `capture_patch` as `ignore`.
|
|
88
|
+
|
|
89
|
+
Call at the end of `setup`, before the agent runs, and keep the result in host
|
|
90
|
+
memory beside `resolve_head`'s SHA. Whatever it lists came with the image, so the
|
|
91
|
+
agent cannot be credited with it, and a patch that carries it fails `git apply` in
|
|
92
|
+
a fresh container of that same image — which is what an isolated grading box is.
|
|
93
|
+
|
|
94
|
+
Sandbox snapshotting will make this free: once runtimes can snapshot and diff a
|
|
95
|
+
filesystem, the pre-agent untracked set falls out of the diff with no setup-side
|
|
96
|
+
bookkeeping in any taskset. Drop this then.
|
|
97
|
+
"""
|
|
98
|
+
result = await runtime.run(
|
|
99
|
+
["sh", "-c", "git ls-files --others --exclude-standard -z"], env or {}
|
|
100
|
+
)
|
|
101
|
+
if result.exit_code != 0:
|
|
102
|
+
return []
|
|
103
|
+
return [path for path in (result.stdout or "").split("\0") if path]
|
|
104
|
+
|
|
105
|
+
|
|
77
106
|
async def capture_patch(
|
|
78
|
-
trace: Trace,
|
|
107
|
+
trace: Trace,
|
|
108
|
+
runtime: Runtime,
|
|
109
|
+
base_commit: str = "",
|
|
110
|
+
env: dict | None = None,
|
|
111
|
+
write_path: str | None = None,
|
|
112
|
+
ignore: list[str] | None = None,
|
|
79
113
|
) -> None:
|
|
80
114
|
"""Snapshot the agent's cumulative diff into `trace.info["patch"]`.
|
|
81
115
|
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
116
|
+
`ignore` names paths to leave out — pass `snapshot_untracked`'s list from setup, or
|
|
117
|
+
`git add -A` credits the agent with untracked files the image shipped. R2E-Gym boxes
|
|
118
|
+
ship three (`datasets`, `install.sh`, `run_tests.sh`), and a patch carrying them
|
|
119
|
+
fails `git apply` in a fresh container of that very image — which is what an
|
|
120
|
+
isolated grading box is.
|
|
121
|
+
|
|
122
|
+
Two failure modes, attributed differently, because they deserve different outcomes.
|
|
123
|
+
|
|
124
|
+
A non-zero exit means the box answered and git refused: a stale `index.lock` from a
|
|
125
|
+
killed agent command, a deleted `.git`, a `base_commit` the agent rewrote out of
|
|
126
|
+
existence, a disk it filled. That is the agent's own environment, so it records
|
|
127
|
+
`info["patch_error"]` and lets the rollout score — a run with no patch grades as a
|
|
128
|
+
run that changed nothing, which is the right reward.
|
|
129
|
+
|
|
130
|
+
An exception means the box never answered: the sandbox died, the exec timed out,
|
|
131
|
+
the transport dropped. Nothing there is the policy's doing, so it raises. Scoring
|
|
132
|
+
the rollout anyway would feed a zero to training that says only that our
|
|
133
|
+
infrastructure failed.
|
|
134
|
+
|
|
135
|
+
`write_path` additionally writes the patch to that path inside the box, for tasks
|
|
136
|
+
graded in a second sandbox. Point it at `vf.ARTIFACTS_DIR` (e.g.
|
|
137
|
+
`/logs/artifacts/patch.diff`) and collection picks it up with no declaration. Leave
|
|
138
|
+
it on the convention sweep rather than declaring it as an `Artifact`: a declared
|
|
139
|
+
path is collected strictly, which would turn an agent-broken repo into a rollout
|
|
140
|
+
error instead of the low score it should earn.
|
|
85
141
|
"""
|
|
86
142
|
nonce = uuid.uuid4().hex
|
|
87
143
|
full, capped = f"{_FULL}_{nonce}", f"{_CAPPED}_{nonce}"
|
|
88
144
|
cmd = _DIFF.format(full=full, capped=capped, cap=PATCH_CAP_BYTES + 1)
|
|
89
145
|
try:
|
|
90
146
|
result = await runtime.run(
|
|
91
|
-
["sh", "-c", cmd],
|
|
147
|
+
["sh", "-c", cmd, "vf-capture-patch", *(ignore or [])],
|
|
92
148
|
{**(env or {}), "VF_DIFF_BASE": base_commit or "HEAD"},
|
|
93
149
|
)
|
|
94
150
|
if result.exit_code != 0:
|
|
151
|
+
# Not every runtime raises when the box is gone — Docker returns `docker
|
|
152
|
+
# exec`'s own non-zero result, which is indistinguishable from git failing.
|
|
153
|
+
# One probe on the failure path tells the two apart before we blame anyone.
|
|
154
|
+
if (await runtime.run(["true"], {})).exit_code != 0:
|
|
155
|
+
raise SandboxError(
|
|
156
|
+
f"patch capture failed and the box stopped answering: "
|
|
157
|
+
f"{(result.stderr or '').strip()[-300:]}"
|
|
158
|
+
)
|
|
95
159
|
trace.info["patch_error"] = (
|
|
96
160
|
f"exit={result.exit_code} {(result.stderr or '').strip()[-500:]}"
|
|
97
161
|
)
|
|
98
162
|
return
|
|
99
163
|
raw = await runtime.read(capped)
|
|
100
|
-
except Exception as exc: # noqa: BLE001 - capture must never fail the rollout
|
|
101
|
-
trace.info["patch_error"] = f"{type(exc).__name__}: {exc}"
|
|
102
|
-
return
|
|
103
164
|
finally:
|
|
104
165
|
# Unique names don't overwrite each other, so leftovers would accumulate
|
|
105
166
|
# on shared-filesystem runtimes; removal is best-effort by design.
|
|
@@ -111,3 +172,7 @@ async def capture_patch(
|
|
|
111
172
|
raw = raw[:PATCH_CAP_BYTES]
|
|
112
173
|
trace.info["patch_truncated"] = True
|
|
113
174
|
trace.info["patch"] = raw.decode("utf-8", errors="replace")
|
|
175
|
+
if write_path is not None:
|
|
176
|
+
parent = str(PurePosixPath(write_path).parent)
|
|
177
|
+
await runtime.run(["mkdir", "-p", parent], env or {})
|
|
178
|
+
await runtime.write(write_path, raw)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.dev53
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -167,8 +167,9 @@ verifiers/utils/threaded_sandbox_client.py,sha256=XRzYhWBpMev7kw_aNMWAy6lfa-9uAn
|
|
|
167
167
|
verifiers/utils/tool_utils.py,sha256=gInWZODWQUZN4TyEMuuITyOUm9qlHJxtcvr1zZl_zJs,1020
|
|
168
168
|
verifiers/utils/usage_utils.py,sha256=Mw6WC78r6EsLyRFdP9ICveqptkjizbSkAC86ZH0geVU,3918
|
|
169
169
|
verifiers/utils/version_utils.py,sha256=39gV8-mNF9GspAgKloLo8oy7_MVluMrwGPrntQv2hx4,2656
|
|
170
|
-
verifiers/v1/__init__.py,sha256=
|
|
170
|
+
verifiers/v1/__init__.py,sha256=uqq1Jzek51cYa2TyHqzDwU6zlOmD3wKLF063Kjh_iN0,7521
|
|
171
171
|
verifiers/v1/agent.py,sha256=dj4GabQSs6Wzwbph5Rlx-efsVQsSuWV2YI5hfvq5mGA,31990
|
|
172
|
+
verifiers/v1/artifacts.py,sha256=G-YTfTAYA35zxKzM-jK60TprndfQevPA65hWDw6ae7Y,5877
|
|
172
173
|
verifiers/v1/decorators.py,sha256=9CcCwgZlRjVGPqT5G1CFJUHP4OIMuBGkUJZkcI6dnbk,3865
|
|
173
174
|
verifiers/v1/env.py,sha256=m-1TNzs9YyFbMkoW6c54vI5xG4jnW3LvLgvJhOgbg04,18447
|
|
174
175
|
verifiers/v1/episode.py,sha256=KWyM9ovTphEIo2260BYu2HvtnOOTxjLRDPrF_jTCiuA,2497
|
|
@@ -183,8 +184,8 @@ verifiers/v1/retries.py,sha256=quUGVruUkZx-9WQcFiocMfL_M056voU8gwDqsuLR3f4,5412
|
|
|
183
184
|
verifiers/v1/rollout.py,sha256=GJzHgJFlw8O1vDtWM__2JhSiG2GqbfjtH2ri1MRfw3g,20495
|
|
184
185
|
verifiers/v1/scoring.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
|
|
185
186
|
verifiers/v1/session.py,sha256=J6ZPuvsg6vMTJAs9dMUMtfdjCuvp6Hom877DVdKFKlY,6912
|
|
186
|
-
verifiers/v1/state.py,sha256=
|
|
187
|
-
verifiers/v1/task.py,sha256=
|
|
187
|
+
verifiers/v1/state.py,sha256=mcQBl9ItRPVrcVr7YzHLbEqKdDSfybhtWD8XDEkeui0,717
|
|
188
|
+
verifiers/v1/task.py,sha256=Ofd3Sh8bfFeEy6nhgeHV0CT2-9lAoGypQyVYhtSP1oc,12206
|
|
188
189
|
verifiers/v1/taskset.py,sha256=RPFkpRkeEyPX557_mK-JwraBuUKsguiCUKSFyotTclU,4636
|
|
189
190
|
verifiers/v1/trace.py,sha256=9KqBeGaRUtWh6Gpdf6G1c7UZso0DB9Mjmtkzl6I_oYE,19458
|
|
190
191
|
verifiers/v1/types.py,sha256=5tZyG4r4bLJ9a14oybHy7bSCpt5X5A17eRwmIszuRo8,9118
|
|
@@ -237,8 +238,8 @@ verifiers/v1/dialects/base.py,sha256=YZQDnasQseDXK-GHLfJLZgyHIIygS80YhD_sTjmUbfA
|
|
|
237
238
|
verifiers/v1/dialects/chat.py,sha256=R_otqi5N0snB1QVKIYloNeLHR9kc7OmI06HYg1ir_Lw,14021
|
|
238
239
|
verifiers/v1/dialects/responses.py,sha256=ismnxUikPcPUIm80FUTPBbj-Cn1e0yGAsISMEseAWRc,13679
|
|
239
240
|
verifiers/v1/envs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
240
|
-
verifiers/v1/envs/agentic_judge/__init__.py,sha256=
|
|
241
|
-
verifiers/v1/envs/agentic_judge/env.py,sha256=
|
|
241
|
+
verifiers/v1/envs/agentic_judge/__init__.py,sha256=6vwtMCQ_cuvuubCF5-nrE5W7gGgzE-LIPKorkM5DXdw,309
|
|
242
|
+
verifiers/v1/envs/agentic_judge/env.py,sha256=NEv1AUbOrGSFDJNzMVmAUkNVk89L1Cpyu36Ndrg3xFI,17003
|
|
242
243
|
verifiers/v1/envs/best_of_n/__init__.py,sha256=mubASiwXMNIhnQFDccNAc7yZ0OFsOVK8hH8WIx9b2A4,119
|
|
243
244
|
verifiers/v1/envs/best_of_n/env.py,sha256=KEUJBCVqblnXP_lEwm3mjGdaikBMkecz8UjOFPBhxc0,1987
|
|
244
245
|
verifiers/v1/envs/single_agent/__init__.py,sha256=r0edAc_6gtHBhPzeBGVcdHZZqSdCPZnLMje6eVgyz5U,138
|
|
@@ -308,7 +309,7 @@ verifiers/v1/serve/server.py,sha256=VBOD3Gzq4p0ZD_8xgtG039i3JaefhOH0MmR6cxG56iY,
|
|
|
308
309
|
verifiers/v1/serve/types.py,sha256=y8Cf9wKOZRKDXcThFWVygTFHITJofOinGiIzwjQGCUE,2674
|
|
309
310
|
verifiers/v1/tasksets/__init__.py,sha256=b62O4WDNbGfJH_4NjYmznn5otWovdfvUgJZyoIZyLh8,596
|
|
310
311
|
verifiers/v1/tasksets/harbor/__init__.py,sha256=wwVGA8N7E2Q2qA5WGHuLRrn4DgG5oaRHrCS2RW3NXVY,195
|
|
311
|
-
verifiers/v1/tasksets/harbor/taskset.py,sha256=
|
|
312
|
+
verifiers/v1/tasksets/harbor/taskset.py,sha256=pkXYS1vknI66SXQV86ypUEZ4kCBdVH2QjMuGjWt8UC0,18283
|
|
312
313
|
verifiers/v1/tasksets/lean/__init__.py,sha256=oyv-GbJQBvCyGwo7Ysxm2NjZW-aX4j1uDoHU0f-E7xQ,767
|
|
313
314
|
verifiers/v1/tasksets/lean/scoring.py,sha256=sfzsT6MUK0zdMgAQ57_P_0CBaPHy3M7go71QMr5hDrs,10599
|
|
314
315
|
verifiers/v1/tasksets/lean/taskset.py,sha256=Fcn4UgTcdjMYnJ4CThaFZ28Rz_QDx0e21vJdW_vp12g,9019
|
|
@@ -321,7 +322,7 @@ verifiers/v1/utils/aio.py,sha256=0yNFOqB2oO6Qktc6NI_ZSAGuo1aHsIK_wuT1BtmoZ3I,153
|
|
|
321
322
|
verifiers/v1/utils/compile.py,sha256=plpZYt-UD8q7Yx5pD7Rfy1w7Jer5TCBk2CJGHiNcPio,5696
|
|
322
323
|
verifiers/v1/utils/format.py,sha256=NfQo9M5KMzWCZpQaet5YtsJx56vCKAZPfbjZZJLJhhM,2187
|
|
323
324
|
verifiers/v1/utils/generic.py,sha256=2VIN9RapMO9pVkFHB1UKUHck1vlTJ1pjfad2NzxuuXA,2055
|
|
324
|
-
verifiers/v1/utils/git.py,sha256=
|
|
325
|
+
verifiers/v1/utils/git.py,sha256=MTRinOvG4EddfjdahHkFZUihHmhJ37n5ROtgjzjE1NA,8366
|
|
325
326
|
verifiers/v1/utils/image.py,sha256=OFw_wdwVtbdwa8uJ3X3vYfhpUAi0CTH5HMBupjVsRRQ,282
|
|
326
327
|
verifiers/v1/utils/install.py,sha256=fWNsyKrw_PyhC0Qhqv5Ri0adFm5Ddx4AVy_hbsra1GU,1390
|
|
327
328
|
verifiers/v1/utils/interrupt.py,sha256=F-KKhc5ndPJJfhd3SuMqyqhXhA32FhCRy5KWFJpEoM4,1179
|
|
@@ -329,8 +330,8 @@ verifiers/v1/utils/logging.py,sha256=OcMHA6NsYux3oIzjPuI95rDWmFBHNcHDjZeNIhXTX-Y
|
|
|
329
330
|
verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
|
|
330
331
|
verifiers/v1/utils/sampling.py,sha256=52TJ5-Hmpvp4YSgKNR-LmAPVwm4RDcH-o2-BnbVTmYQ,965
|
|
331
332
|
verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
|
|
332
|
-
verifiers-0.2.2.
|
|
333
|
-
verifiers-0.2.2.
|
|
334
|
-
verifiers-0.2.2.
|
|
335
|
-
verifiers-0.2.2.
|
|
336
|
-
verifiers-0.2.2.
|
|
333
|
+
verifiers-0.2.2.dev53.dist-info/METADATA,sha256=vvfSCIqaWIr6egUCiFZ505i0TPbN5Cc0boYECWJ-sMg,4545
|
|
334
|
+
verifiers-0.2.2.dev53.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
335
|
+
verifiers-0.2.2.dev53.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
|
|
336
|
+
verifiers-0.2.2.dev53.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
337
|
+
verifiers-0.2.2.dev53.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|