verifiers 0.2.2.dev74__py3-none-any.whl → 0.2.2.dev76__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/__init__.py +2 -2
- verifiers/v1/agent.py +2 -8
- verifiers/v1/configs/cli/__init__.py +1 -2
- verifiers/v1/configs/cli/eval.py +2 -2
- verifiers/v1/configs/serve.py +5 -5
- verifiers/v1/runtimes/__init__.py +19 -0
- verifiers/v1/runtimes/base.py +36 -2
- verifiers/v1/runtimes/docker/__init__.py +1 -1
- verifiers/v1/runtimes/modal.py +1 -1
- verifiers/v1/runtimes/prime.py +1 -1
- verifiers/v1/runtimes/subprocess.py +1 -1
- verifiers/v1/tasksets/harbor/__init__.py +9 -1
- verifiers/v1/tasksets/harbor/env.py +124 -0
- verifiers/v1/tasksets/harbor/taskset.py +247 -51
- {verifiers-0.2.2.dev74.dist-info → verifiers-0.2.2.dev76.dist-info}/METADATA +1 -1
- {verifiers-0.2.2.dev74.dist-info → verifiers-0.2.2.dev76.dist-info}/RECORD +19 -20
- {verifiers-0.2.2.dev74.dist-info → verifiers-0.2.2.dev76.dist-info}/entry_points.txt +0 -1
- verifiers/v1/cli/serve.py +0 -79
- verifiers/v1/configs/cli/serve.py +0 -62
- {verifiers-0.2.2.dev74.dist-info → verifiers-0.2.2.dev76.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev74.dist-info → verifiers-0.2.2.dev76.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/__init__.py
CHANGED
|
@@ -22,7 +22,7 @@ from verifiers.v1.configs.legacy import LegacyEnvConfig
|
|
|
22
22
|
from verifiers.v1.configs.retries import RetryConfig
|
|
23
23
|
from verifiers.v1.configs.serve import (
|
|
24
24
|
ElasticPoolConfig,
|
|
25
|
-
|
|
25
|
+
ServeConfig,
|
|
26
26
|
StaticPoolConfig,
|
|
27
27
|
pool_serve_kwargs,
|
|
28
28
|
)
|
|
@@ -254,7 +254,7 @@ __all__ = [ # noqa: RUF022 - grouped by public API area
|
|
|
254
254
|
"Env",
|
|
255
255
|
"SingleAgentEnv",
|
|
256
256
|
"EnvConfig",
|
|
257
|
-
"
|
|
257
|
+
"ServeConfig",
|
|
258
258
|
"LegacyEnvConfig",
|
|
259
259
|
"resolve_env_field",
|
|
260
260
|
"narrowed_env_annotation",
|
verifiers/v1/agent.py
CHANGED
|
@@ -29,7 +29,7 @@ from verifiers.v1.runtimes import (
|
|
|
29
29
|
Runtime,
|
|
30
30
|
RuntimeConfig,
|
|
31
31
|
SubprocessConfig,
|
|
32
|
-
|
|
32
|
+
provision_runtime,
|
|
33
33
|
runtime_is_local,
|
|
34
34
|
)
|
|
35
35
|
from verifiers.v1.session import RolloutLimits
|
|
@@ -547,14 +547,8 @@ class Agent:
|
|
|
547
547
|
if task is not None
|
|
548
548
|
else self.runtime_config
|
|
549
549
|
)
|
|
550
|
-
|
|
551
|
-
try:
|
|
552
|
-
# start() inside the try: a failed start may already hold a remote
|
|
553
|
-
# sandbox, so it must reach stop() (safe on a partially-started runtime).
|
|
554
|
-
await runtime.start()
|
|
550
|
+
async with provision_runtime(config) as runtime:
|
|
555
551
|
yield runtime
|
|
556
|
-
finally:
|
|
557
|
-
await runtime.stop()
|
|
558
552
|
|
|
559
553
|
|
|
560
554
|
class _EpisodeAgent(Agent):
|
|
@@ -3,7 +3,6 @@
|
|
|
3
3
|
from verifiers.v1.configs.cli.debug import DebugConfig
|
|
4
4
|
from verifiers.v1.configs.cli.eval import EvalConfig
|
|
5
5
|
from verifiers.v1.configs.cli.init import InitConfig
|
|
6
|
-
from verifiers.v1.configs.cli.serve import ServeConfig
|
|
7
6
|
from verifiers.v1.configs.cli.validate import ValidateConfig
|
|
8
7
|
|
|
9
|
-
__all__ = ["DebugConfig", "EvalConfig", "InitConfig", "
|
|
8
|
+
__all__ = ["DebugConfig", "EvalConfig", "InitConfig", "ValidateConfig"]
|
verifiers/v1/configs/cli/eval.py
CHANGED
|
@@ -10,7 +10,7 @@ from verifiers.v1.clients import ClientConfig, EvalClientConfig
|
|
|
10
10
|
from verifiers.v1.configs.cli.env import narrowed_env_annotation, resolve_env_field
|
|
11
11
|
from verifiers.v1.configs.env import EnvConfig
|
|
12
12
|
from verifiers.v1.configs.legacy import LegacyEnvConfig
|
|
13
|
-
from verifiers.v1.configs.serve import
|
|
13
|
+
from verifiers.v1.configs.serve import ServeConfig
|
|
14
14
|
from verifiers.v1.envs.single_agent import SingleAgentEnvConfig
|
|
15
15
|
from verifiers.v1.types import SamplingConfig
|
|
16
16
|
|
|
@@ -19,7 +19,7 @@ class EvalConfig(BaseConfig):
|
|
|
19
19
|
env: SerializeAsAny[EnvConfig] = SingleAgentEnvConfig()
|
|
20
20
|
"""The environment — which env, its seed taskset, each agent, its knobs. Narrowed to
|
|
21
21
|
the selected env's config class by the env id, else the taskset id."""
|
|
22
|
-
serve:
|
|
22
|
+
serve: ServeConfig = ServeConfig()
|
|
23
23
|
"""How the env is hosted under `--server`: the worker pool, each worker's episode
|
|
24
24
|
bound. Ignored by an in-process run."""
|
|
25
25
|
legacy: LegacyEnvConfig = LegacyEnvConfig()
|
verifiers/v1/configs/serve.py
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
"""How an env is served: the worker pool, where it binds, and each worker's bound.
|
|
2
2
|
|
|
3
3
|
Serving is its own axis. `[env]` says what runs; this says how it's hosted, so an
|
|
4
|
-
eval
|
|
5
|
-
|
|
4
|
+
eval and a trainer's env server configure it identically instead of each flattening
|
|
5
|
+
pool knobs onto its own config."""
|
|
6
6
|
|
|
7
7
|
from typing import Annotated, Literal
|
|
8
8
|
|
|
@@ -35,10 +35,10 @@ PoolConfig = Annotated[
|
|
|
35
35
|
]
|
|
36
36
|
|
|
37
37
|
|
|
38
|
-
class
|
|
38
|
+
class ServeConfig(BaseConfig):
|
|
39
39
|
"""The `[serve]` block: the worker pool, the ZMQ address, and each worker's
|
|
40
|
-
episode bound. Read by whoever hosts the env —
|
|
41
|
-
|
|
40
|
+
episode bound. Read by whoever hosts the env — a server-backed eval, a
|
|
41
|
+
trainer's env-server process."""
|
|
42
42
|
|
|
43
43
|
pool: PoolConfig = Field(default_factory=ElasticPoolConfig)
|
|
44
44
|
"""Worker-pool sizing. `elastic` (default) starts at one worker and scales up on
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
from collections.abc import AsyncIterator
|
|
2
|
+
from contextlib import asynccontextmanager
|
|
1
3
|
from typing import Annotated
|
|
2
4
|
|
|
3
5
|
from pydantic import Field
|
|
@@ -45,6 +47,22 @@ def make_runtime(config: RuntimeConfig, name: str | None = None) -> Runtime:
|
|
|
45
47
|
return runtime
|
|
46
48
|
|
|
47
49
|
|
|
50
|
+
@asynccontextmanager
|
|
51
|
+
async def provision_runtime(
|
|
52
|
+
config: RuntimeConfig, name: str | None = None
|
|
53
|
+
) -> AsyncIterator[Runtime]:
|
|
54
|
+
"""Provision a box from `config` and tear it down on exit.
|
|
55
|
+
|
|
56
|
+
`start()` sits inside the `try`: a failed start may already hold a paid sandbox, so
|
|
57
|
+
it has to reach `stop()` (which is safe on a partially-started runtime)."""
|
|
58
|
+
runtime = make_runtime(config, name)
|
|
59
|
+
try:
|
|
60
|
+
await runtime.start()
|
|
61
|
+
yield runtime
|
|
62
|
+
finally:
|
|
63
|
+
await runtime.stop()
|
|
64
|
+
|
|
65
|
+
|
|
48
66
|
def runtime_is_local(config: RuntimeConfig) -> bool:
|
|
49
67
|
"""Whether a runtime of this config exchanges host-local URLs without a public
|
|
50
68
|
tunnel, read off the runtime class without provisioning one."""
|
|
@@ -71,5 +89,6 @@ __all__ = [
|
|
|
71
89
|
"SubprocessRuntime",
|
|
72
90
|
"SubprocessRuntimeInfo",
|
|
73
91
|
"make_runtime",
|
|
92
|
+
"provision_runtime",
|
|
74
93
|
"runtime_is_local",
|
|
75
94
|
]
|
verifiers/v1/runtimes/base.py
CHANGED
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
import asyncio
|
|
4
4
|
import atexit
|
|
5
|
+
import base64
|
|
5
6
|
import contextlib
|
|
6
7
|
import hashlib
|
|
7
8
|
import logging
|
|
@@ -285,9 +286,42 @@ class Runtime(ABC):
|
|
|
285
286
|
argv = await self.prepare_uv_script(script, env)
|
|
286
287
|
return await self.run([*argv, *(args or [])], env or {})
|
|
287
288
|
|
|
289
|
+
async def read(self, path: str, max_bytes: int | None = None) -> bytes:
|
|
290
|
+
"""Read `path` into host memory. `max_bytes` caps the transfer, raising past
|
|
291
|
+
the cap — for a file written by something we don't control, whose size we
|
|
292
|
+
can't assume. The cap is enforced inside the box rather than after the
|
|
293
|
+
transfer, and base64 because `run` returns decoded text. Framework method —
|
|
294
|
+
override `_read`, not this."""
|
|
295
|
+
if max_bytes is None:
|
|
296
|
+
return await self._read(path)
|
|
297
|
+
# Through a temp file, not a pipe: `head | base64` exits with base64's 0
|
|
298
|
+
# even when the path is missing, and a missing file must raise here just
|
|
299
|
+
# as it does from `_read`.
|
|
300
|
+
result = await self.run(
|
|
301
|
+
[
|
|
302
|
+
"sh",
|
|
303
|
+
"-c",
|
|
304
|
+
(
|
|
305
|
+
"t=$(mktemp) || exit 1; "
|
|
306
|
+
'head -c "$1" -- "$2" > "$t" || { rm -f "$t"; exit 1; }; '
|
|
307
|
+
'base64 < "$t"; rc=$?; rm -f "$t"; exit $rc'
|
|
308
|
+
),
|
|
309
|
+
"sh",
|
|
310
|
+
str(max_bytes + 1),
|
|
311
|
+
path,
|
|
312
|
+
],
|
|
313
|
+
{},
|
|
314
|
+
)
|
|
315
|
+
if result.exit_code:
|
|
316
|
+
raise SandboxError(f"read {path!r}: {result.stderr.strip()[-500:]}")
|
|
317
|
+
data = base64.b64decode(result.stdout)
|
|
318
|
+
if len(data) > max_bytes:
|
|
319
|
+
raise SandboxError(f"read {path!r}: over the {max_bytes} byte limit")
|
|
320
|
+
return data
|
|
321
|
+
|
|
288
322
|
@abstractmethod
|
|
289
|
-
async def
|
|
290
|
-
|
|
323
|
+
async def _read(self, path: str) -> bytes:
|
|
324
|
+
"""Read the whole file at `path`; `read` adds the optional transfer cap."""
|
|
291
325
|
|
|
292
326
|
@abstractmethod
|
|
293
327
|
async def write(self, path: str, data: bytes) -> None:
|
|
@@ -334,7 +334,7 @@ class DockerRuntime(Runtime):
|
|
|
334
334
|
if run.exit_code != 0:
|
|
335
335
|
raise SandboxError(f"docker exec -d failed: {run.stderr.strip()}")
|
|
336
336
|
|
|
337
|
-
async def
|
|
337
|
+
async def _read(self, path: str) -> bytes:
|
|
338
338
|
proc = await asyncio.create_subprocess_exec(
|
|
339
339
|
"docker",
|
|
340
340
|
"exec",
|
verifiers/v1/runtimes/modal.py
CHANGED
|
@@ -165,7 +165,7 @@ class ModalRuntime(Runtime):
|
|
|
165
165
|
return path
|
|
166
166
|
return f"{self.config.workdir.rstrip('/')}/{path}"
|
|
167
167
|
|
|
168
|
-
async def
|
|
168
|
+
async def _read(self, path: str) -> bytes:
|
|
169
169
|
try:
|
|
170
170
|
return await self._sandbox.filesystem.read_bytes.aio(self._abs(path))
|
|
171
171
|
except Exception as e:
|
verifiers/v1/runtimes/prime.py
CHANGED
|
@@ -270,7 +270,7 @@ class PrimeRuntime(Runtime):
|
|
|
270
270
|
f"prime background launch failed: {result.stderr.strip()}"
|
|
271
271
|
)
|
|
272
272
|
|
|
273
|
-
async def
|
|
273
|
+
async def _read(self, path: str) -> bytes:
|
|
274
274
|
# Avoid background-job log limits and base64 overhead by downloading binary data directly.
|
|
275
275
|
# The temporary file is removed on every exit, and its byte read stays off the event loop.
|
|
276
276
|
target = (
|
|
@@ -96,7 +96,7 @@ class SubprocessRuntime(Runtime):
|
|
|
96
96
|
proc
|
|
97
97
|
) # killed in stop() — a host process won't die on its own
|
|
98
98
|
|
|
99
|
-
async def
|
|
99
|
+
async def _read(self, path: str) -> bytes:
|
|
100
100
|
return await asyncio.to_thread((self.workdir / path).read_bytes)
|
|
101
101
|
|
|
102
102
|
async def write(self, path: str, data: bytes) -> None:
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
from verifiers.v1.tasksets.harbor.env import HarborEnv, HarborEnvConfig
|
|
1
2
|
from verifiers.v1.tasksets.harbor.taskset import (
|
|
2
3
|
HarborConfig,
|
|
3
4
|
HarborData,
|
|
@@ -5,4 +6,11 @@ from verifiers.v1.tasksets.harbor.taskset import (
|
|
|
5
6
|
HarborTaskset,
|
|
6
7
|
)
|
|
7
8
|
|
|
8
|
-
__all__ = [
|
|
9
|
+
__all__ = [
|
|
10
|
+
"HarborConfig",
|
|
11
|
+
"HarborData",
|
|
12
|
+
"HarborEnv",
|
|
13
|
+
"HarborEnvConfig",
|
|
14
|
+
"HarborTask",
|
|
15
|
+
"HarborTaskset",
|
|
16
|
+
]
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""The harbor taskset's own env: the single solver seat, plus separate-verifier
|
|
2
|
+
grading for tasks that declare ``[verifier].environment_mode = "separate"``.
|
|
3
|
+
|
|
4
|
+
The default env for harbor runs (the taskset package exports it). A shared-verifier
|
|
5
|
+
task runs exactly as under the single-agent env: one `agent` trace, graded in the
|
|
6
|
+
box it worked in. A separate-verifier task is graded by `finalize` instead: the
|
|
7
|
+
solver's declared artifacts travel (collected by its task `finalize` while its box
|
|
8
|
+
is alive), a fresh box is provisioned from the task's verifier declaration,
|
|
9
|
+
`tests/` is staged there, and the verifier's rewards land on the solver's trace.
|
|
10
|
+
No second agent is involved — the verifier is the task's own `tests/test.sh`.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import asyncio
|
|
14
|
+
import logging
|
|
15
|
+
from contextlib import AsyncExitStack
|
|
16
|
+
|
|
17
|
+
from pydantic import Field
|
|
18
|
+
|
|
19
|
+
import verifiers.v1 as vf
|
|
20
|
+
from verifiers.v1.runtimes import RuntimeConfig, provision_runtime
|
|
21
|
+
from verifiers.v1.tasksets.harbor.taskset import (
|
|
22
|
+
HarborTask,
|
|
23
|
+
verifier_box_data,
|
|
24
|
+
)
|
|
25
|
+
from verifiers.v1.utils.artifacts import restore
|
|
26
|
+
from verifiers.v1.utils.compile import resolve_runtime_config
|
|
27
|
+
from verifiers.v1.utils.retries import backoff
|
|
28
|
+
|
|
29
|
+
logger = logging.getLogger(__name__)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class HarborEnvConfig(vf.EnvConfig):
|
|
33
|
+
agent: vf.AgentConfig = vf.AgentConfig()
|
|
34
|
+
"""The one seat — the policy under evaluation/training; pin
|
|
35
|
+
`--env.agent.harness.*` to choose its program or runtime."""
|
|
36
|
+
verifier_runtime: RuntimeConfig | None = None
|
|
37
|
+
"""Where a separate-verifier task grades. None derives the grading box from
|
|
38
|
+
the solver's runtime policy; set it (e.g. `--env.verifier-runtime.type prime
|
|
39
|
+
--env.verifier-runtime.vm true`) when the verifier needs different placement
|
|
40
|
+
than the agent."""
|
|
41
|
+
verifier_retries: int = Field(2, ge=0)
|
|
42
|
+
"""Extra attempts at provisioning-and-grading the separate box before the
|
|
43
|
+
episode fails. Grading is deterministic; what these retry is the
|
|
44
|
+
infrastructure around it (image pulls, provisioning)."""
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class HarborEnv(vf.Env[HarborEnvConfig]):
|
|
48
|
+
async def run(self, task: vf.Task, agents: vf.Agents) -> None:
|
|
49
|
+
if not isinstance(task, HarborTask):
|
|
50
|
+
raise TypeError(
|
|
51
|
+
f"the harbor env runs harbor tasks; got {type(task).__name__}"
|
|
52
|
+
)
|
|
53
|
+
if task.data.verifier is None:
|
|
54
|
+
await agents.agent.run(task)
|
|
55
|
+
return
|
|
56
|
+
# Resolve the verifier's box before the solve, so an impossible pairing
|
|
57
|
+
# (e.g. a restricted Prime verifier without vm=true) costs nothing
|
|
58
|
+
# rather than a full agent run.
|
|
59
|
+
self._verifier_config(task)
|
|
60
|
+
await agents.agent.run(task.graded_elsewhere())
|
|
61
|
+
|
|
62
|
+
def _verifier_config(self, task: HarborTask) -> RuntimeConfig:
|
|
63
|
+
base = (
|
|
64
|
+
self.config.verifier_runtime
|
|
65
|
+
if self.config.verifier_runtime is not None
|
|
66
|
+
else self.config.agent.runtime
|
|
67
|
+
)
|
|
68
|
+
return resolve_runtime_config(base, HarborTask(verifier_box_data(task.data)))
|
|
69
|
+
|
|
70
|
+
async def finalize(self, task: vf.Task, episode: vf.Episode) -> None:
|
|
71
|
+
"""Grade a separate-verifier task in its own box, onto the solver's trace.
|
|
72
|
+
|
|
73
|
+
Provision a fresh box from the task's verifier declaration, restore the
|
|
74
|
+
solver's collected artifacts, stage `tests/`, run the verifier, and record
|
|
75
|
+
its rewards (and any extra reward.json keys as metrics) on the solver's
|
|
76
|
+
trace. Infrastructure failures retry per `verifier_retries`; the last one
|
|
77
|
+
fails the episode — a grading box that can't be reached must never read
|
|
78
|
+
as reward 0."""
|
|
79
|
+
if not isinstance(task, HarborTask) or task.data.verifier is None:
|
|
80
|
+
return
|
|
81
|
+
solution = episode.traces[0]
|
|
82
|
+
if not solution.ok:
|
|
83
|
+
return
|
|
84
|
+
grader = HarborTask(verifier_box_data(task.data))
|
|
85
|
+
scores = await self._grade(self._verifier_config(task), grader, solution)
|
|
86
|
+
items = scores.items() if isinstance(scores, dict) else [("solved", scores)]
|
|
87
|
+
for name, value in items:
|
|
88
|
+
solution.record_reward(name, value)
|
|
89
|
+
|
|
90
|
+
async def _grade(
|
|
91
|
+
self, config: RuntimeConfig, grader: HarborTask, solution: vf.Trace
|
|
92
|
+
) -> float | dict[str, float]:
|
|
93
|
+
last: Exception | None = None
|
|
94
|
+
for attempt in range(self.config.verifier_retries + 1):
|
|
95
|
+
if attempt:
|
|
96
|
+
delay = backoff(attempt - 1)
|
|
97
|
+
logger.warning(
|
|
98
|
+
"harbor verifier attempt %d/%d failed (%s); retrying in %.1fs",
|
|
99
|
+
attempt,
|
|
100
|
+
self.config.verifier_retries + 1,
|
|
101
|
+
last,
|
|
102
|
+
delay,
|
|
103
|
+
)
|
|
104
|
+
await asyncio.sleep(delay)
|
|
105
|
+
try:
|
|
106
|
+
# The scoring deadline covers provisioning and grading, but not the
|
|
107
|
+
# box's teardown: a score already in hand must not be discarded
|
|
108
|
+
# because the teardown ran out the clock.
|
|
109
|
+
async with AsyncExitStack() as boxes:
|
|
110
|
+
async with asyncio.timeout(grader.data.timeout.scoring):
|
|
111
|
+
box = await boxes.enter_async_context(provision_runtime(config))
|
|
112
|
+
await box.prepare_setup()
|
|
113
|
+
# Artifacts first, tests second: an artifact entry pointing
|
|
114
|
+
# into /tests must not survive staging, which wipes and
|
|
115
|
+
# rebuilds that directory.
|
|
116
|
+
await restore(box, solution.state.artifacts)
|
|
117
|
+
await grader._stage_tests(box, wipe=True)
|
|
118
|
+
await box.prepare_execution([])
|
|
119
|
+
scores = await grader._graded(box, solution)
|
|
120
|
+
return scores
|
|
121
|
+
except Exception as e: # noqa: BLE001 - each attempt's failure is retried
|
|
122
|
+
last = e
|
|
123
|
+
assert last is not None
|
|
124
|
+
raise last
|
|
@@ -1,18 +1,25 @@
|
|
|
1
1
|
"""Harbor tasksets backed by Harbor Hub packages.
|
|
2
2
|
|
|
3
3
|
The Harbor CLI downloads and caches each task directory. Its verifier runs in the
|
|
4
|
-
|
|
4
|
+
runtime the harness edited — or, when the task asks for it with
|
|
5
|
+
``[verifier].environment_mode = "separate"``, in a second box the agent never
|
|
6
|
+
touched, carrying only what the task declared — the harbor env provisions and
|
|
7
|
+
grades that box (see ``env.py``). Either way the score lands in
|
|
5
8
|
``/logs/verifier/reward.json`` or the legacy ``reward.txt``.
|
|
6
9
|
|
|
7
10
|
A pullable ``[environment].docker_image`` becomes ``TaskData.image``. Verifiers does
|
|
8
11
|
not build Dockerfile-only environments, so those are rejected unless ``ignore_dockerfile``
|
|
9
12
|
deliberately uses the harness runtime image. Tasks without an environment also use that
|
|
10
|
-
image unless ``require_image`` is set.
|
|
13
|
+
image unless ``require_image`` is set. The same rule applies to a declared
|
|
14
|
+
``[verifier.environment]``: it needs a pullable ``docker_image``, since Harbor would
|
|
15
|
+
otherwise build the verifier image from ``tests/Dockerfile``.
|
|
11
16
|
"""
|
|
12
17
|
|
|
13
18
|
import asyncio
|
|
19
|
+
import copy
|
|
14
20
|
import hashlib
|
|
15
21
|
import io
|
|
22
|
+
import logging
|
|
16
23
|
import shutil
|
|
17
24
|
import subprocess
|
|
18
25
|
import sys
|
|
@@ -26,7 +33,7 @@ from typing import Annotated
|
|
|
26
33
|
from pydantic import BaseModel, ConfigDict, Field, TypeAdapter, ValidationError
|
|
27
34
|
|
|
28
35
|
from verifiers.v1.configs.taskset import TasksetConfig
|
|
29
|
-
from verifiers.v1.errors import SandboxError
|
|
36
|
+
from verifiers.v1.errors import SandboxError, TaskError
|
|
30
37
|
from verifiers.v1.runtimes import Runtime
|
|
31
38
|
from verifiers.v1.task import Task, TaskData, TaskResources, TaskTimeout
|
|
32
39
|
from verifiers.v1.taskset import Taskset
|
|
@@ -34,9 +41,12 @@ from verifiers.v1.trace import Trace
|
|
|
34
41
|
from verifiers.v1.utils.artifacts import Artifact, collect
|
|
35
42
|
from verifiers.v1.utils.decorators import reward
|
|
36
43
|
|
|
44
|
+
logger = logging.getLogger(__name__)
|
|
45
|
+
|
|
37
46
|
CACHE = Path.home() / ".cache" / "harbor"
|
|
38
47
|
HARBOR_INSTALL_HINT = "uv sync --python 3.12 --extra harbor"
|
|
39
48
|
REWARD_JSON = "/logs/verifier/reward.json"
|
|
49
|
+
MAX_REWARD_BYTES = 1024 * 1024
|
|
40
50
|
REWARD_JSON_ADAPTER = TypeAdapter(
|
|
41
51
|
float | Annotated[dict[str, float], Field(min_length=1)],
|
|
42
52
|
config=ConfigDict(strict=True, allow_inf_nan=False),
|
|
@@ -76,6 +86,11 @@ class HarborConfig(TasksetConfig):
|
|
|
76
86
|
instead of rejecting it. The Dockerfile is NOT built, so the task scores against the
|
|
77
87
|
harness image rather than its declared environment — only correct when that image already
|
|
78
88
|
has what the task needs (e.g. you've pointed the runtime at the right image)."""
|
|
89
|
+
ignore_separate_verifier: bool = False
|
|
90
|
+
"""Grade every task in the agent's own box, even one whose `[verifier]` asks for a
|
|
91
|
+
separate one. Trades the isolation for a sandbox per task; useful when provisioning
|
|
92
|
+
is the bottleneck. Note what it gives up: the grader becomes reachable by the agent
|
|
93
|
+
that just ran."""
|
|
79
94
|
|
|
80
95
|
|
|
81
96
|
class Author(BaseModel):
|
|
@@ -90,6 +105,27 @@ class CollectHook(BaseModel):
|
|
|
90
105
|
timeout_sec: float = 600.0
|
|
91
106
|
|
|
92
107
|
|
|
108
|
+
class VerifierConfig(BaseModel):
|
|
109
|
+
"""The box this task's verifier wants, when it wants one of its own.
|
|
110
|
+
|
|
111
|
+
`None` on `HarborData` means shared — grade where the agent worked, which is still
|
|
112
|
+
Harbor's default and every task that says nothing."""
|
|
113
|
+
|
|
114
|
+
image: str | None = None
|
|
115
|
+
"""Pullable ref from `[verifier.environment].docker_image`. None keeps the task's
|
|
116
|
+
own image, which is what Harbor's fresh copy of `[environment]` resolves to."""
|
|
117
|
+
resources: TaskResources = TaskResources()
|
|
118
|
+
workdir: str | None = None
|
|
119
|
+
fresh_copy: bool = False
|
|
120
|
+
"""Whether this came from Harbor's fresh copy of `[environment]` rather than a
|
|
121
|
+
declared `[verifier.environment]`. A fresh copy inherits the agent box's resolved
|
|
122
|
+
resources; a declared environment states its own, and what it omits falls back to
|
|
123
|
+
the run's rather than to the agent's task-derived values."""
|
|
124
|
+
network_allow: list[str] = Field(default_factory=lambda: ["*"])
|
|
125
|
+
"""Destinations the verifier may reach, from the verifier's network mode. `["*"]`
|
|
126
|
+
is unrestricted; `[]` is Harbor's `no-network` / `allow_internet = false`."""
|
|
127
|
+
|
|
128
|
+
|
|
93
129
|
class HarborData(TaskData):
|
|
94
130
|
"""Parsed ``task.toml`` metadata plus the host-side verifier directory.
|
|
95
131
|
|
|
@@ -111,6 +147,9 @@ class HarborData(TaskData):
|
|
|
111
147
|
collect: list[CollectHook] = Field(default_factory=list)
|
|
112
148
|
"""`[[verifier.collect]]` blocks: commands that snapshot runtime state into files
|
|
113
149
|
after the agent stops, so the files can travel to a grading box as artifacts."""
|
|
150
|
+
verifier: VerifierConfig | None = None
|
|
151
|
+
"""The verifier's own box, when `[verifier].environment_mode` asks for one. None
|
|
152
|
+
grades in the agent's box."""
|
|
114
153
|
|
|
115
154
|
|
|
116
155
|
class HarborTask(Task[HarborData]):
|
|
@@ -145,30 +184,59 @@ class HarborTask(Task[HarborData]):
|
|
|
145
184
|
)
|
|
146
185
|
trace.state.artifacts = await collect(runtime, self.data.artifacts)
|
|
147
186
|
|
|
148
|
-
|
|
149
|
-
|
|
187
|
+
def graded_elsewhere(self) -> "HarborTask":
|
|
188
|
+
"""A copy whose `solved` records nothing here: the harbor env grades this
|
|
189
|
+
task's finished work in a separate box of the task's choosing."""
|
|
190
|
+
clone = copy.copy(self)
|
|
191
|
+
clone._graded_elsewhere = True
|
|
192
|
+
return clone
|
|
193
|
+
|
|
194
|
+
_graded_elsewhere: bool = False
|
|
195
|
+
|
|
196
|
+
async def _stage_tests(self, runtime: Runtime, wipe: bool = False) -> None:
|
|
197
|
+
"""Put the task package's `tests/` in `/tests`, where `test.sh` expects it.
|
|
198
|
+
|
|
199
|
+
Raises rather than scoring stale state: a leftover reward file — planted by
|
|
200
|
+
the agent or shipped in the image — must be gone before `test.sh` runs, so a
|
|
201
|
+
removal that fails must not fall through to reading it.
|
|
202
|
+
|
|
203
|
+
`wipe` for a box we did not watch being built: a fresh container of the task's
|
|
204
|
+
image can ship its own `/tests`, and a leftover file there would be graded as
|
|
205
|
+
though it came from the package.
|
|
206
|
+
"""
|
|
150
207
|
await runtime.write(
|
|
151
208
|
"/tmp/tests.tgz", make_tar(Path(self.data.task_dir) / "tests")
|
|
152
209
|
)
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
"mkdir -p /logs/verifier /tests && tar -xzf /tmp/tests.tgz -C /tests",
|
|
158
|
-
],
|
|
159
|
-
{},
|
|
160
|
-
)
|
|
161
|
-
await runtime.run(
|
|
162
|
-
[
|
|
163
|
-
"sh",
|
|
164
|
-
"-c",
|
|
165
|
-
(
|
|
166
|
-
"rm -f /logs/verifier/reward.json /logs/verifier/reward.txt"
|
|
167
|
-
" && cd /tests && bash test.sh"
|
|
168
|
-
),
|
|
169
|
-
],
|
|
170
|
-
verifier_env(self.data),
|
|
210
|
+
stage = (
|
|
211
|
+
f"{'rm -rf /tests && ' if wipe else ''}"
|
|
212
|
+
"rm -f /logs/verifier/reward.json /logs/verifier/reward.txt && "
|
|
213
|
+
"mkdir -p /logs/verifier /tests && tar -xzf /tmp/tests.tgz -C /tests"
|
|
171
214
|
)
|
|
215
|
+
result = await runtime.run(["sh", "-c", stage], {})
|
|
216
|
+
if result.exit_code:
|
|
217
|
+
raise TaskError(
|
|
218
|
+
f"staging tests failed (exit {result.exit_code}): "
|
|
219
|
+
f"{(result.stderr or result.stdout).strip()[-500:]}"
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
@reward(weight=1.0)
|
|
223
|
+
async def solved(self, runtime: Runtime, trace: Trace) -> float | dict[str, float]:
|
|
224
|
+
if self.data.verifier is not None:
|
|
225
|
+
if not self._graded_elsewhere:
|
|
226
|
+
raise TaskError(
|
|
227
|
+
f"task {self.data.name!r} declares a separate verifier "
|
|
228
|
+
'([verifier].environment_mode = "separate"); grade it through '
|
|
229
|
+
"the harbor env (this taskset's default), or force shared "
|
|
230
|
+
"grading with --taskset.ignore-separate-verifier"
|
|
231
|
+
)
|
|
232
|
+
return {}
|
|
233
|
+
await self._stage_tests(runtime)
|
|
234
|
+
return await self._graded(runtime, trace)
|
|
235
|
+
|
|
236
|
+
async def _graded(self, runtime: Runtime, trace: Trace) -> float | dict[str, float]:
|
|
237
|
+
# By absolute path, in the runtime's configured workdir: Harbor execs the
|
|
238
|
+
# script the same way, and scripts do grade the agent's work at `$PWD`.
|
|
239
|
+
await runtime.run(["bash", "/tests/test.sh"], verifier_env(self.data))
|
|
172
240
|
scores = await self._reward_json(runtime)
|
|
173
241
|
if scores is not None:
|
|
174
242
|
if isinstance(scores, dict) and "reward" in scores:
|
|
@@ -178,19 +246,75 @@ class HarborTask(Task[HarborData]):
|
|
|
178
246
|
return {"reward": scores["reward"]}
|
|
179
247
|
return scores
|
|
180
248
|
try:
|
|
181
|
-
reward = (
|
|
249
|
+
reward = (
|
|
250
|
+
(
|
|
251
|
+
await runtime.read(
|
|
252
|
+
"/logs/verifier/reward.txt", max_bytes=MAX_REWARD_BYTES
|
|
253
|
+
)
|
|
254
|
+
)
|
|
255
|
+
.decode()
|
|
256
|
+
.strip()
|
|
257
|
+
)
|
|
182
258
|
return float(reward or 0)
|
|
183
259
|
except (SandboxError, OSError, ValueError):
|
|
184
260
|
return 0.0
|
|
185
261
|
|
|
186
262
|
async def _reward_json(self, runtime: Runtime) -> float | dict[str, float] | None:
|
|
187
|
-
"""Read Harbor's scalar or keyed JSON reward, if it is valid.
|
|
263
|
+
"""Read Harbor's scalar or keyed JSON reward, if it is valid.
|
|
264
|
+
|
|
265
|
+
Bounded: this is a grading input, and nothing guarantees its size.
|
|
266
|
+
"""
|
|
188
267
|
try:
|
|
189
|
-
return REWARD_JSON_ADAPTER.validate_json(
|
|
268
|
+
return REWARD_JSON_ADAPTER.validate_json(
|
|
269
|
+
await runtime.read(REWARD_JSON, max_bytes=MAX_REWARD_BYTES)
|
|
270
|
+
)
|
|
190
271
|
except (SandboxError, OSError, ValidationError):
|
|
191
272
|
return None
|
|
192
273
|
|
|
193
274
|
|
|
275
|
+
def verifier_box_data(data: HarborData) -> HarborData:
|
|
276
|
+
"""The verifier's box, declared as task data — the harbor env resolves the
|
|
277
|
+
grading runtime from it (image, workdir, resources, network policy), exactly
|
|
278
|
+
as the solver's box resolves from the solver task's.
|
|
279
|
+
|
|
280
|
+
Which box follows Harbor: a declared `[verifier.environment]` states its own
|
|
281
|
+
image, workdir, and resources, and what it omits is the run's default; a
|
|
282
|
+
fresh copy of `[environment]` keeps the task's own. The verifier's network
|
|
283
|
+
policy applies either way."""
|
|
284
|
+
verifier = data.verifier
|
|
285
|
+
if verifier is None:
|
|
286
|
+
raise TaskError(f"task {data.name!r} declares no separate verifier")
|
|
287
|
+
fresh = verifier.fresh_copy
|
|
288
|
+
return data.model_copy(
|
|
289
|
+
update={
|
|
290
|
+
"name": f"{data.name} (verifier)",
|
|
291
|
+
"image": verifier.image if verifier.image is not None else data.image,
|
|
292
|
+
"workdir": data.workdir if fresh else verifier.workdir,
|
|
293
|
+
"resources": data.resources if fresh else verifier.resources,
|
|
294
|
+
"network_allow": list(verifier.network_allow),
|
|
295
|
+
"network_block": [],
|
|
296
|
+
}
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def task_resources(environment, multiplier: float) -> TaskResources:
|
|
301
|
+
"""Harbor environment resource requests, scaled, as `TaskResources`.
|
|
302
|
+
|
|
303
|
+
Harbor declares CPU counts and MB sizes; `TaskResources` wants counts and GB.
|
|
304
|
+
GPU requests are never scaled.
|
|
305
|
+
"""
|
|
306
|
+
return TaskResources(
|
|
307
|
+
cpu=environment.cpus * multiplier if environment.cpus else None,
|
|
308
|
+
memory=environment.memory_mb / 1024 * multiplier
|
|
309
|
+
if environment.memory_mb
|
|
310
|
+
else None,
|
|
311
|
+
gpu=str(environment.gpus) if environment.gpus else None,
|
|
312
|
+
disk=environment.storage_mb / 1024 * multiplier
|
|
313
|
+
if environment.storage_mb
|
|
314
|
+
else None,
|
|
315
|
+
)
|
|
316
|
+
|
|
317
|
+
|
|
194
318
|
def harbor_cli() -> str:
|
|
195
319
|
scripts_dir = Path(sys.executable).parent
|
|
196
320
|
harbor_bin = shutil.which("harbor", path=str(scripts_dir))
|
|
@@ -318,7 +442,7 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
|
|
|
318
442
|
|
|
319
443
|
harbor_task = HarborModelTask(task_dir)
|
|
320
444
|
parsed = harbor_task.config
|
|
321
|
-
artifacts,
|
|
445
|
+
artifacts, hooks, verifier = parse_verifier_extras(task_dir, parsed, harbor_config)
|
|
322
446
|
environment = parsed.environment
|
|
323
447
|
network = parsed.agent.explicit_phase_policy() or environment.resolve_baseline()
|
|
324
448
|
task, meta = parsed.task, parsed.metadata
|
|
@@ -368,18 +492,7 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
|
|
|
368
492
|
if scoring_timeout is not None
|
|
369
493
|
else None,
|
|
370
494
|
),
|
|
371
|
-
resources=
|
|
372
|
-
cpu=environment.cpus * harbor_config.resource_multiplier
|
|
373
|
-
if environment.cpus
|
|
374
|
-
else None,
|
|
375
|
-
memory=environment.memory_mb / 1024 * harbor_config.resource_multiplier
|
|
376
|
-
if environment.memory_mb
|
|
377
|
-
else None,
|
|
378
|
-
gpu=str(environment.gpus) if environment.gpus else None,
|
|
379
|
-
disk=environment.storage_mb / 1024 * harbor_config.resource_multiplier
|
|
380
|
-
if environment.storage_mb
|
|
381
|
-
else None,
|
|
382
|
-
),
|
|
495
|
+
resources=task_resources(environment, harbor_config.resource_multiplier),
|
|
383
496
|
keywords=task.keywords if task else [],
|
|
384
497
|
authors=authors,
|
|
385
498
|
difficulty=meta.get("difficulty"),
|
|
@@ -388,14 +501,22 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
|
|
|
388
501
|
task_dir=str(task_dir),
|
|
389
502
|
verifier_env=parsed.verifier.env,
|
|
390
503
|
artifacts=artifacts,
|
|
391
|
-
collect=
|
|
504
|
+
collect=hooks,
|
|
505
|
+
verifier=verifier,
|
|
392
506
|
)
|
|
393
507
|
|
|
394
508
|
|
|
395
509
|
def parse_verifier_extras(
|
|
396
|
-
task_dir: Path, parsed
|
|
397
|
-
) -> tuple[list[Artifact], list[CollectHook]]:
|
|
398
|
-
"""
|
|
510
|
+
task_dir: Path, parsed, harbor_config: HarborConfig
|
|
511
|
+
) -> tuple[list[Artifact], list[CollectHook], VerifierConfig | None]:
|
|
512
|
+
"""Harbor's `artifacts`, `[[verifier.collect]]` blocks, and verifier environment,
|
|
513
|
+
narrowed to what verifiers' verifier-runtime integration can honor.
|
|
514
|
+
|
|
515
|
+
The convention dir is deliberately not prepended here (Harbor's
|
|
516
|
+
`with_convention_entry` would): collection injects it itself, as an optional sweep.
|
|
517
|
+
Prepending it would make it an explicitly declared entry, and declared entries are
|
|
518
|
+
required — which would fail every task that never writes there.
|
|
519
|
+
"""
|
|
399
520
|
from harbor.constants import MAIN_SERVICE_NAME
|
|
400
521
|
from harbor.models.task.artifacts import (
|
|
401
522
|
effective_artifact_service,
|
|
@@ -403,13 +524,6 @@ def parse_verifier_extras(
|
|
|
403
524
|
)
|
|
404
525
|
|
|
405
526
|
verifier = parsed.verifier
|
|
406
|
-
if verifier.environment is not None:
|
|
407
|
-
raise ValueError(
|
|
408
|
-
f"{task_dir.name}: [verifier.environment] declares a separate verifier "
|
|
409
|
-
"image. Grading runs in a fresh box built from the task's own image, so "
|
|
410
|
-
"only the agent's delta has to travel; a different verifier image needs "
|
|
411
|
-
"the full working tree copied over and isn't supported yet."
|
|
412
|
-
)
|
|
413
527
|
if verifier.user is not None:
|
|
414
528
|
raise ValueError(f"{task_dir.name}: [verifier].user is not supported")
|
|
415
529
|
|
|
@@ -447,7 +561,89 @@ def parse_verifier_extras(
|
|
|
447
561
|
)
|
|
448
562
|
hooks.append(CollectHook(command=hook.command, timeout_sec=hook.timeout_sec))
|
|
449
563
|
|
|
450
|
-
return artifacts, hooks
|
|
564
|
+
return artifacts, hooks, parse_verifier_environment(task_dir, parsed, harbor_config)
|
|
565
|
+
|
|
566
|
+
|
|
567
|
+
def parse_verifier_environment(
|
|
568
|
+
task_dir: Path, parsed, harbor_config: HarborConfig
|
|
569
|
+
) -> VerifierConfig | None:
|
|
570
|
+
"""The box Harbor wants this task's verifier in, or None to grade in the agent's.
|
|
571
|
+
|
|
572
|
+
Harbor resolves `[verifier.environment]` if declared, else a deep copy of
|
|
573
|
+
`[environment]` — so a mode-only `separate` lands on the task's own image and needs
|
|
574
|
+
nothing but a second box. A declared environment is the case that can name a
|
|
575
|
+
different image, and the case that can name none at all: there Harbor builds
|
|
576
|
+
`tests/Dockerfile`, which verifiers never does.
|
|
577
|
+
"""
|
|
578
|
+
from harbor.models.task.config import NetworkMode, TaskOS
|
|
579
|
+
from harbor.models.task.verifier_mode import (
|
|
580
|
+
VerifierEnvironmentMode,
|
|
581
|
+
resolve_effective_verifier_env_config,
|
|
582
|
+
resolve_task_verifier_mode,
|
|
583
|
+
)
|
|
584
|
+
|
|
585
|
+
if resolve_task_verifier_mode(parsed) != VerifierEnvironmentMode.SEPARATE:
|
|
586
|
+
return None
|
|
587
|
+
if harbor_config.ignore_separate_verifier:
|
|
588
|
+
logger.warning(
|
|
589
|
+
"%s: asks for a separate verifier; grading in the agent's box anyway "
|
|
590
|
+
"(--taskset.ignore-separate-verifier)",
|
|
591
|
+
task_dir.name,
|
|
592
|
+
)
|
|
593
|
+
return None
|
|
594
|
+
|
|
595
|
+
environment = resolve_effective_verifier_env_config(parsed, None)
|
|
596
|
+
if environment is None: # unreachable while the mode is SEPARATE
|
|
597
|
+
raise ValueError(f"{task_dir.name}: separate verifier resolved no environment")
|
|
598
|
+
declared = parsed.verifier.environment is not None
|
|
599
|
+
|
|
600
|
+
if declared and environment.docker_image is None:
|
|
601
|
+
if not harbor_config.ignore_dockerfile:
|
|
602
|
+
raise ValueError(
|
|
603
|
+
f"{task_dir.name}: [verifier.environment] names no docker_image, so "
|
|
604
|
+
"Harbor would build the verifier image from tests/Dockerfile. Verifiers "
|
|
605
|
+
"pulls images and never builds them: build and push it yourself (e.g. "
|
|
606
|
+
"`prime images push`) and set [verifier.environment].docker_image to the "
|
|
607
|
+
"resulting ref, or pass --taskset.ignore-dockerfile to grade in the "
|
|
608
|
+
"agent's image instead."
|
|
609
|
+
)
|
|
610
|
+
logger.warning(
|
|
611
|
+
"%s: [verifier.environment] names no docker_image — grading in the agent's "
|
|
612
|
+
"image rather than building tests/Dockerfile, so the verifier runs somewhere "
|
|
613
|
+
"the task never declared",
|
|
614
|
+
task_dir.name,
|
|
615
|
+
)
|
|
616
|
+
unsupported = [
|
|
617
|
+
field
|
|
618
|
+
for field in ("healthcheck", "mcp_servers", "skills_dir", "gpu_types", "tpu")
|
|
619
|
+
if getattr(environment, field, None)
|
|
620
|
+
]
|
|
621
|
+
if environment.os != TaskOS.LINUX or unsupported:
|
|
622
|
+
raise ValueError(
|
|
623
|
+
f"{task_dir.name}: verifier environment declares "
|
|
624
|
+
f"{unsupported or environment.os}, which verifiers' verifier-runtime "
|
|
625
|
+
"integration cannot honor"
|
|
626
|
+
)
|
|
627
|
+
|
|
628
|
+
network = parsed.verifier.explicit_phase_policy() or environment.resolve_baseline()
|
|
629
|
+
return VerifierConfig(
|
|
630
|
+
image=environment.docker_image if declared else None,
|
|
631
|
+
# A declared environment states its own resources; what it leaves out is the
|
|
632
|
+
# run's default, not the agent task's. A fresh copy is the task's environment,
|
|
633
|
+
# so it keeps whatever the agent box resolved to.
|
|
634
|
+
resources=(
|
|
635
|
+
task_resources(environment, harbor_config.resource_multiplier)
|
|
636
|
+
if declared
|
|
637
|
+
else TaskResources()
|
|
638
|
+
),
|
|
639
|
+
workdir=environment.workdir if declared else None,
|
|
640
|
+
fresh_copy=not declared,
|
|
641
|
+
network_allow=(
|
|
642
|
+
["*"]
|
|
643
|
+
if network.network_mode == NetworkMode.PUBLIC
|
|
644
|
+
else list(network.allowed_hosts)
|
|
645
|
+
),
|
|
646
|
+
)
|
|
451
647
|
|
|
452
648
|
|
|
453
649
|
def verifier_env(task: HarborData) -> dict[str, str]:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.dev76
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -167,8 +167,8 @@ verifiers/utils/threaded_sandbox_client.py,sha256=XRzYhWBpMev7kw_aNMWAy6lfa-9uAn
|
|
|
167
167
|
verifiers/utils/tool_utils.py,sha256=gInWZODWQUZN4TyEMuuITyOUm9qlHJxtcvr1zZl_zJs,1020
|
|
168
168
|
verifiers/utils/usage_utils.py,sha256=Mw6WC78r6EsLyRFdP9ICveqptkjizbSkAC86ZH0geVU,3918
|
|
169
169
|
verifiers/utils/version_utils.py,sha256=39gV8-mNF9GspAgKloLo8oy7_MVluMrwGPrntQv2hx4,2656
|
|
170
|
-
verifiers/v1/__init__.py,sha256=
|
|
171
|
-
verifiers/v1/agent.py,sha256=
|
|
170
|
+
verifiers/v1/__init__.py,sha256=v6qrJQFeNL2b4wd0PsDWiituw2Y-f5rMaFmdbdP6SdM,7505
|
|
171
|
+
verifiers/v1/agent.py,sha256=MQVKGjUhrP8uzRXAq94uJHkEsdHFziPxQHZ7TOD324I,31306
|
|
172
172
|
verifiers/v1/env.py,sha256=g6fpoG-Z9lM3vTKrrU-QRJY3S88_y9zsPHwBvfs5vdY,17865
|
|
173
173
|
verifiers/v1/episode.py,sha256=YtuRTV_PpwDglXFeN6UNg8IswpP1HseFLCYtascGMNI,3218
|
|
174
174
|
verifiers/v1/errors.py,sha256=kQeEPX06TwAIuiz7MlWFYirrpKbx5STi6A5b_dtQQoA,6885
|
|
@@ -192,7 +192,6 @@ verifiers/v1/cli/init.py,sha256=Qj96Kwzfhaejhrc0yYqHNRl3ncoID028RGlHvS1C98Q,8508
|
|
|
192
192
|
verifiers/v1/cli/output.py,sha256=pMS5Z0EBgyAVShNCAk9jVURNeLOaTfeRWpt0J8hHjq8,5890
|
|
193
193
|
verifiers/v1/cli/replay.py,sha256=DAh9kB5PB-ppZLGhxdJwrCPch_Yky5HKhplwGFhKEz0,10164
|
|
194
194
|
verifiers/v1/cli/resolve.py,sha256=OsNw4C3r9xOVFa-mminJFcex_bAsJ6EQ8K7A_H8u2kQ,4370
|
|
195
|
-
verifiers/v1/cli/serve.py,sha256=_mdJaHVVe4pXzcpAot0ugcmiN3ra2gS4GrGmLghnX5M,2866
|
|
196
195
|
verifiers/v1/cli/validate.py,sha256=7ax6FqIzNBSYjd3JXK4v7Vbq34Rozyh1OvGpYnGqlSg,10266
|
|
197
196
|
verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
|
|
198
197
|
verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
|
|
@@ -216,16 +215,15 @@ verifiers/v1/configs/harness.py,sha256=CG8hGbHl3rr6yawH3YYMAFl01M7Piiwf4w5oiw2Va
|
|
|
216
215
|
verifiers/v1/configs/judge.py,sha256=O7oe28tS24fKRXUML_-6vHrVGdG6w7061SPQXOH7XvM,2223
|
|
217
216
|
verifiers/v1/configs/legacy.py,sha256=YOJM6L5u6-jYpZ1CLpPInufKdmz1lEsMNKIRdT2JaaY,1080
|
|
218
217
|
verifiers/v1/configs/retries.py,sha256=-nPlgmz7J_NZ4GOYG6oCc3thXo4La2SufBmvQGvvPGo,821
|
|
219
|
-
verifiers/v1/configs/serve.py,sha256=
|
|
218
|
+
verifiers/v1/configs/serve.py,sha256=nSoJcTX2S0Vy9EUeNqW0Yo-f9_MUqpBiY-Q9FF_4cRo,2576
|
|
220
219
|
verifiers/v1/configs/task.py,sha256=1ozZjfpahin7I_TqzC7f3xr7CHSthAdydegFPXW4d9k,1101
|
|
221
220
|
verifiers/v1/configs/taskset.py,sha256=o-tr1xkmGmYG10_ySG-Du_djWS-dN4h7cv14WSvJ4eY,851
|
|
222
|
-
verifiers/v1/configs/cli/__init__.py,sha256=
|
|
221
|
+
verifiers/v1/configs/cli/__init__.py,sha256=TJaWQQHrcyXoeqS7P9S25ztd78YJ30_9_Ug0czRH7Tw,374
|
|
223
222
|
verifiers/v1/configs/cli/debug.py,sha256=VFsZ1tiKGvvx5mFDOzqqeU4cRtQkJ8OeD8mmRBZhy8w,2849
|
|
224
223
|
verifiers/v1/configs/cli/env.py,sha256=DJbJy0-DqqtmFm0r7NpvRBbuxXnXc1o-yurDKUBSGJk,2701
|
|
225
|
-
verifiers/v1/configs/cli/eval.py,sha256=
|
|
224
|
+
verifiers/v1/configs/cli/eval.py,sha256=HqIBtt8DuZEHZ07KgtE1jiqZy8OsDMfTahTuBAvslcA,6863
|
|
226
225
|
verifiers/v1/configs/cli/init.py,sha256=tMaTDF_lU1Af3JmQpksk2mwe8CAfpnHsquXhwepZhYo,1019
|
|
227
226
|
verifiers/v1/configs/cli/replay.py,sha256=xvMDZzIkvjgYl-a3ragjXI65DsD6FYl6gEFM5qpdIuA,2913
|
|
228
|
-
verifiers/v1/configs/cli/serve.py,sha256=1rAt0QOL-XT3DMmRUIA40t1E3YuoD1vOADfsiTwFBf0,2981
|
|
229
227
|
verifiers/v1/configs/cli/validate.py,sha256=UdgC6bQ2RYx3t8n9HjsImAHV_ifoOxYOSlbtdZzOnZY,2298
|
|
230
228
|
verifiers/v1/dialects/__init__.py,sha256=p72C7VYrFQzVgTcfmzpHGSwXlvThfc4pvjBxsqw7kHE,786
|
|
231
229
|
verifiers/v1/dialects/anthropic.py,sha256=nj9J7OOhexmPHa-sWiNP2xlI9_bGhvwM5VWzdh19pBc,13526
|
|
@@ -297,13 +295,13 @@ verifiers/v1/mcp/__init__.py,sha256=pm8ZWBiL1q1tY4wHvJzM_QrZ9oR8aGm-bIRWiRJsgHI,
|
|
|
297
295
|
verifiers/v1/mcp/launch.py,sha256=wBpyeiOO86l1H2lVMbLVSV9Lbp90XeyXqauuK8snfEU,20171
|
|
298
296
|
verifiers/v1/mcp/server.py,sha256=dRBxytFgS6rv1BS6t5EgBd3xeF6U1nTyt0ArDCLk3VE,12301
|
|
299
297
|
verifiers/v1/mcp/toolset.py,sha256=T8Ioiq4wphxVOTHUDf28ZoQBM2ZYEevGnGgw8TFEMiA,1002
|
|
300
|
-
verifiers/v1/runtimes/__init__.py,sha256=
|
|
301
|
-
verifiers/v1/runtimes/base.py,sha256=
|
|
298
|
+
verifiers/v1/runtimes/__init__.py,sha256=Kwe612RjtGzrQhc19QdXc7CmZAnQLH-EdK_xqkKP3Tk,2629
|
|
299
|
+
verifiers/v1/runtimes/base.py,sha256=tZRELSfZmi3l_1BsjabnNC3mEPfqNvUNevdnRvLPv6w,16257
|
|
302
300
|
verifiers/v1/runtimes/limiters.py,sha256=ZjjYUFEt20yH8ggL5uU9w9U500AOB44qSUTQ7bYyh0c,2730
|
|
303
|
-
verifiers/v1/runtimes/modal.py,sha256=
|
|
304
|
-
verifiers/v1/runtimes/prime.py,sha256=
|
|
305
|
-
verifiers/v1/runtimes/subprocess.py,sha256=
|
|
306
|
-
verifiers/v1/runtimes/docker/__init__.py,sha256=
|
|
301
|
+
verifiers/v1/runtimes/modal.py,sha256=eFiT-Ovf_RwQZbSw3Dwv2EXeHV0d6uGvPswqKYKCw5k,8826
|
|
302
|
+
verifiers/v1/runtimes/prime.py,sha256=kwvPqUI8D-GvcFwYRdpTHE1hXS0vfJGPlFOLgzm4Mvs,14756
|
|
303
|
+
verifiers/v1/runtimes/subprocess.py,sha256=ZrHRlTuTyCrcSKpiPD2ElcPpe97EvVsXJ8BfwUaNXiI,5988
|
|
304
|
+
verifiers/v1/runtimes/docker/__init__.py,sha256=vyEgf6XgAwjGgBicJchJvNgcDgYXrRvyQC4W52-23GU,14556
|
|
307
305
|
verifiers/v1/runtimes/docker/egress.py,sha256=37UXgNX9gnoLxjMg4KZaJIGoApCAM6MK22eJGNoMi38,14514
|
|
308
306
|
verifiers/v1/serve/__init__.py,sha256=sr_zkDzAHnOcQm-QRm8xwJBmCMFD-xhKIFPaDZvMbUs,641
|
|
309
307
|
verifiers/v1/serve/client.py,sha256=CxpOwCn4ehuCXfqDJeV9zj7O0XC7MPcI-OZ1L_CSDlo,7057
|
|
@@ -311,8 +309,9 @@ verifiers/v1/serve/pool.py,sha256=AyOMjDYi9PLGA8Z-bnr6Fd5cnubuRavEBuudbVWtq2E,14
|
|
|
311
309
|
verifiers/v1/serve/server.py,sha256=eFWJsTUDbMywKQXKDuk57QvS7Y4IZwW2Ok3LwJHMFLI,9200
|
|
312
310
|
verifiers/v1/serve/types.py,sha256=3F4IIGIQgYITlEFxeX3uNJYrJlKugFrPTozbDNwDfBY,2674
|
|
313
311
|
verifiers/v1/tasksets/__init__.py,sha256=b62O4WDNbGfJH_4NjYmznn5otWovdfvUgJZyoIZyLh8,596
|
|
314
|
-
verifiers/v1/tasksets/harbor/__init__.py,sha256=
|
|
315
|
-
verifiers/v1/tasksets/harbor/
|
|
312
|
+
verifiers/v1/tasksets/harbor/__init__.py,sha256=JxThzHFYEkuSfQbVf3pEnjordgdfe4MU-QOsSSbyovU,326
|
|
313
|
+
verifiers/v1/tasksets/harbor/env.py,sha256=bNF5dfbhcsdrd_8JmW5ved8Yk-mHhrOgKkVEA563qEo,5847
|
|
314
|
+
verifiers/v1/tasksets/harbor/taskset.py,sha256=vmDjP7dhxchZ5-J9IUeiATwwchxUvAeciCevD6W_N4c,29032
|
|
316
315
|
verifiers/v1/tasksets/lean/__init__.py,sha256=oyv-GbJQBvCyGwo7Ysxm2NjZW-aX4j1uDoHU0f-E7xQ,767
|
|
317
316
|
verifiers/v1/tasksets/lean/scoring.py,sha256=sfzsT6MUK0zdMgAQ57_P_0CBaPHy3M7go71QMr5hDrs,10599
|
|
318
317
|
verifiers/v1/tasksets/lean/taskset.py,sha256=XgUqX0uJvHC8sKFhbK92iu6nUGcPuXCUpI3ZBhDcb7c,9025
|
|
@@ -338,8 +337,8 @@ verifiers/v1/utils/platform.py,sha256=56Ixmk1SER6q5LvzyYcA0hgzGpzhxKZL8Hiqp4c_XV
|
|
|
338
337
|
verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
|
|
339
338
|
verifiers/v1/utils/score.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
|
|
340
339
|
verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
|
|
341
|
-
verifiers-0.2.2.
|
|
342
|
-
verifiers-0.2.2.
|
|
343
|
-
verifiers-0.2.2.
|
|
344
|
-
verifiers-0.2.2.
|
|
345
|
-
verifiers-0.2.2.
|
|
340
|
+
verifiers-0.2.2.dev76.dist-info/METADATA,sha256=vOCZL_bCXmOR9KHHvbzSCszxDmGtjDRTNxI0I8Zn8Hs,4545
|
|
341
|
+
verifiers-0.2.2.dev76.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
342
|
+
verifiers-0.2.2.dev76.dist-info/entry_points.txt,sha256=v6v0QT9vVExnfn4br42MostI_o3f3ZbWYlTormz-U3g,515
|
|
343
|
+
verifiers-0.2.2.dev76.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
344
|
+
verifiers-0.2.2.dev76.dist-info/RECORD,,
|
|
@@ -4,7 +4,6 @@ eval = verifiers.v1.cli.eval.main:main
|
|
|
4
4
|
gepa = verifiers.v1.cli.gepa:main
|
|
5
5
|
init = verifiers.v1.cli.init:main
|
|
6
6
|
replay = verifiers.v1.cli.replay:main
|
|
7
|
-
serve = verifiers.v1.cli.serve:main
|
|
8
7
|
validate = verifiers.v1.cli.validate:main
|
|
9
8
|
vf-build = verifiers.scripts.build:main
|
|
10
9
|
vf-eval = verifiers.scripts.eval:main
|
verifiers/v1/cli/serve.py
DELETED
|
@@ -1,79 +0,0 @@
|
|
|
1
|
-
"""Environment-server CLI entrypoint."""
|
|
2
|
-
|
|
3
|
-
import sys
|
|
4
|
-
from functools import partial
|
|
5
|
-
|
|
6
|
-
from pydantic_config import cli
|
|
7
|
-
|
|
8
|
-
from verifiers.v1.cli.resolve import (
|
|
9
|
-
extract_id,
|
|
10
|
-
narrow_config,
|
|
11
|
-
plugin_errors,
|
|
12
|
-
references_config_file,
|
|
13
|
-
with_positional_taskset,
|
|
14
|
-
)
|
|
15
|
-
from verifiers.v1.configs.cli.serve import ServeConfig
|
|
16
|
-
from verifiers.v1.configs.serve import pool_serve_kwargs
|
|
17
|
-
from verifiers.v1.serve import serve_env
|
|
18
|
-
from verifiers.v1.utils.logging import setup_logging
|
|
19
|
-
|
|
20
|
-
USAGE = "usage: uv run serve [<taskset-id>] [--env.id <id>] [--legacy.id <env-id> (v0)] [options] [@ file.toml]"
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
def main(argv: list[str] | None = None) -> None:
|
|
24
|
-
argv = with_positional_taskset(list(sys.argv[1:]) if argv is None else list(argv))
|
|
25
|
-
|
|
26
|
-
if not argv or any(arg in ("-h", "--help") for arg in argv):
|
|
27
|
-
print(USAGE)
|
|
28
|
-
sys.argv = [sys.argv[0], "--help"]
|
|
29
|
-
with plugin_errors():
|
|
30
|
-
cli(narrow_config(ServeConfig, argv))
|
|
31
|
-
return
|
|
32
|
-
legacy_id = any(a == "--legacy.id" or a.startswith("--legacy.id=") for a in argv)
|
|
33
|
-
# An env-block flag (or a since-moved flat axis) skips the usage gate so the typed
|
|
34
|
-
# parse renders its did-you-mean instead of a bare usage line.
|
|
35
|
-
typed_axis = any(
|
|
36
|
-
a.startswith(("--env.", "--taskset.", "--harness.", "--serve.")) for a in argv
|
|
37
|
-
)
|
|
38
|
-
if (
|
|
39
|
-
not extract_id(argv, "env.taskset")
|
|
40
|
-
and not legacy_id
|
|
41
|
-
and not references_config_file(argv)
|
|
42
|
-
and not typed_axis
|
|
43
|
-
):
|
|
44
|
-
raise SystemExit(
|
|
45
|
-
USAGE
|
|
46
|
-
) # need a taskset (positional / --env.taskset.id), a v0 --legacy.id, or @ file.toml
|
|
47
|
-
|
|
48
|
-
with plugin_errors():
|
|
49
|
-
config_type = narrow_config(ServeConfig, argv)
|
|
50
|
-
sys.argv = [sys.argv[0], *argv]
|
|
51
|
-
config = cli(config_type)
|
|
52
|
-
if config.dry_run:
|
|
53
|
-
print(config.model_dump_json(indent=2, exclude_none=True))
|
|
54
|
-
return
|
|
55
|
-
level = "DEBUG" if config.verbose else "INFO"
|
|
56
|
-
setup_logging(level)
|
|
57
|
-
|
|
58
|
-
# The pool config decides in-process vs router + worker pool (static or elastic); the
|
|
59
|
-
# frontend speaks the same protocol either way. serve_env owns the SIGTERM teardown.
|
|
60
|
-
# Pool workers are spawned with no logging, so hand serve_env the same setup to apply
|
|
61
|
-
# in each one.
|
|
62
|
-
server_kwargs = (
|
|
63
|
-
{
|
|
64
|
-
"env_id": config.legacy.id,
|
|
65
|
-
"env_args": config.legacy.args,
|
|
66
|
-
"extra_env_kwargs": config.legacy.extra_env_kwargs,
|
|
67
|
-
}
|
|
68
|
-
if config.is_legacy
|
|
69
|
-
# `--serve.max-concurrent` is each v1 worker's episode bound; the legacy
|
|
70
|
-
# bridge has never had one.
|
|
71
|
-
else {"config": config.env, "max_concurrent": config.serve.max_concurrent}
|
|
72
|
-
)
|
|
73
|
-
serve_env(
|
|
74
|
-
**pool_serve_kwargs(config.serve.pool),
|
|
75
|
-
legacy=config.is_legacy,
|
|
76
|
-
address=config.serve.address,
|
|
77
|
-
log_setup=partial(setup_logging, level),
|
|
78
|
-
**server_kwargs,
|
|
79
|
-
)
|
|
@@ -1,62 +0,0 @@
|
|
|
1
|
-
"""Environment-server CLI configuration."""
|
|
2
|
-
|
|
3
|
-
from pydantic import AliasChoices, Field, SerializeAsAny, model_validator
|
|
4
|
-
from pydantic_config import BaseConfig
|
|
5
|
-
|
|
6
|
-
from verifiers.v1.configs.cli.env import narrowed_env_annotation, resolve_env_field
|
|
7
|
-
from verifiers.v1.configs.env import EnvConfig
|
|
8
|
-
from verifiers.v1.configs.legacy import LegacyEnvConfig
|
|
9
|
-
from verifiers.v1.configs.serve import ServingConfig
|
|
10
|
-
from verifiers.v1.envs.single_agent import SingleAgentEnvConfig
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
class ServeConfig(BaseConfig):
|
|
14
|
-
"""`uv run serve`: what to serve (`[env]`, or `[legacy]` for a classic v0 env) and
|
|
15
|
-
how it's hosted (`[serve]`)."""
|
|
16
|
-
|
|
17
|
-
env: SerializeAsAny[EnvConfig] = SingleAgentEnvConfig()
|
|
18
|
-
"""The environment — which env, its seed taskset, each agent, its knobs. Narrowed to
|
|
19
|
-
the selected env's config class by the env id, else the taskset id."""
|
|
20
|
-
serve: ServingConfig = ServingConfig()
|
|
21
|
-
"""How it's served: the worker pool, the bind address, each worker's episode bound."""
|
|
22
|
-
legacy: LegacyEnvConfig = LegacyEnvConfig()
|
|
23
|
-
"""A classic (v0) environment to serve through the bridge instead of `[env]`."""
|
|
24
|
-
verbose: bool = Field(False, validation_alias=AliasChoices("verbose", "v"))
|
|
25
|
-
"""Log at debug level instead of info."""
|
|
26
|
-
dry_run: bool = False
|
|
27
|
-
"""Resolve + validate the config and dump it, then exit."""
|
|
28
|
-
|
|
29
|
-
@property
|
|
30
|
-
def is_legacy(self) -> bool:
|
|
31
|
-
"""Whether this run goes through the v0 bridge: a legacy id and no v1 taskset."""
|
|
32
|
-
return self.legacy.id is not None and not self.env.taskset.id
|
|
33
|
-
|
|
34
|
-
@property
|
|
35
|
-
def env_id(self) -> str:
|
|
36
|
-
"""The run's identifier: the v1 env's, else the v0 env id."""
|
|
37
|
-
return self.env.env_id or self.legacy.id or ""
|
|
38
|
-
|
|
39
|
-
@model_validator(mode="after")
|
|
40
|
-
def _refuse_mixed_run(self):
|
|
41
|
-
# A v0 id next to any v1 env identity leaves one of the two going nowhere, and
|
|
42
|
-
# which one depends on `is_legacy`: a taskset makes it False, so the v0 env never
|
|
43
|
-
# loads; a bare `--env.id` leaves it True, so the v0 env runs under the v1 name.
|
|
44
|
-
if self.legacy.id is None or not self.env.env_id:
|
|
45
|
-
return self
|
|
46
|
-
if self.env.taskset.id:
|
|
47
|
-
raise ValueError(
|
|
48
|
-
f"--legacy.id {self.legacy.id!r} is a classic (v0) env and can't combine "
|
|
49
|
-
f"with the v1 taskset {self.env.taskset.id!r}. Pairing a reusable env with "
|
|
50
|
-
f"a taskset is --env.id {self.legacy.id!r} (TOML: id under [env]); to run "
|
|
51
|
-
"the v0 env instead, drop the taskset."
|
|
52
|
-
)
|
|
53
|
-
raise ValueError(
|
|
54
|
-
f"--legacy.id {self.legacy.id!r} is a classic (v0) env and can't combine with "
|
|
55
|
-
f"the v1 env --env.id {self.env.id!r}: the v0 env is what would run, stamped "
|
|
56
|
-
"with the v1 env's name. Keep whichever one you meant to run."
|
|
57
|
-
)
|
|
58
|
-
|
|
59
|
-
@model_validator(mode="before")
|
|
60
|
-
@classmethod
|
|
61
|
-
def _resolve_env(cls, data):
|
|
62
|
-
return resolve_env_field(data, narrowed_env_annotation(cls))
|
|
File without changes
|
|
File without changes
|