verifiers 0.2.2.dev74__py3-none-any.whl → 0.2.2.dev76__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
verifiers/v1/__init__.py CHANGED
@@ -22,7 +22,7 @@ from verifiers.v1.configs.legacy import LegacyEnvConfig
22
22
  from verifiers.v1.configs.retries import RetryConfig
23
23
  from verifiers.v1.configs.serve import (
24
24
  ElasticPoolConfig,
25
- ServingConfig,
25
+ ServeConfig,
26
26
  StaticPoolConfig,
27
27
  pool_serve_kwargs,
28
28
  )
@@ -254,7 +254,7 @@ __all__ = [ # noqa: RUF022 - grouped by public API area
254
254
  "Env",
255
255
  "SingleAgentEnv",
256
256
  "EnvConfig",
257
- "ServingConfig",
257
+ "ServeConfig",
258
258
  "LegacyEnvConfig",
259
259
  "resolve_env_field",
260
260
  "narrowed_env_annotation",
verifiers/v1/agent.py CHANGED
@@ -29,7 +29,7 @@ from verifiers.v1.runtimes import (
29
29
  Runtime,
30
30
  RuntimeConfig,
31
31
  SubprocessConfig,
32
- make_runtime,
32
+ provision_runtime,
33
33
  runtime_is_local,
34
34
  )
35
35
  from verifiers.v1.session import RolloutLimits
@@ -547,14 +547,8 @@ class Agent:
547
547
  if task is not None
548
548
  else self.runtime_config
549
549
  )
550
- runtime = make_runtime(config)
551
- try:
552
- # start() inside the try: a failed start may already hold a remote
553
- # sandbox, so it must reach stop() (safe on a partially-started runtime).
554
- await runtime.start()
550
+ async with provision_runtime(config) as runtime:
555
551
  yield runtime
556
- finally:
557
- await runtime.stop()
558
552
 
559
553
 
560
554
  class _EpisodeAgent(Agent):
@@ -3,7 +3,6 @@
3
3
  from verifiers.v1.configs.cli.debug import DebugConfig
4
4
  from verifiers.v1.configs.cli.eval import EvalConfig
5
5
  from verifiers.v1.configs.cli.init import InitConfig
6
- from verifiers.v1.configs.cli.serve import ServeConfig
7
6
  from verifiers.v1.configs.cli.validate import ValidateConfig
8
7
 
9
- __all__ = ["DebugConfig", "EvalConfig", "InitConfig", "ServeConfig", "ValidateConfig"]
8
+ __all__ = ["DebugConfig", "EvalConfig", "InitConfig", "ValidateConfig"]
@@ -10,7 +10,7 @@ from verifiers.v1.clients import ClientConfig, EvalClientConfig
10
10
  from verifiers.v1.configs.cli.env import narrowed_env_annotation, resolve_env_field
11
11
  from verifiers.v1.configs.env import EnvConfig
12
12
  from verifiers.v1.configs.legacy import LegacyEnvConfig
13
- from verifiers.v1.configs.serve import ServingConfig
13
+ from verifiers.v1.configs.serve import ServeConfig
14
14
  from verifiers.v1.envs.single_agent import SingleAgentEnvConfig
15
15
  from verifiers.v1.types import SamplingConfig
16
16
 
@@ -19,7 +19,7 @@ class EvalConfig(BaseConfig):
19
19
  env: SerializeAsAny[EnvConfig] = SingleAgentEnvConfig()
20
20
  """The environment — which env, its seed taskset, each agent, its knobs. Narrowed to
21
21
  the selected env's config class by the env id, else the taskset id."""
22
- serve: ServingConfig = ServingConfig()
22
+ serve: ServeConfig = ServeConfig()
23
23
  """How the env is hosted under `--server`: the worker pool, each worker's episode
24
24
  bound. Ignored by an in-process run."""
25
25
  legacy: LegacyEnvConfig = LegacyEnvConfig()
@@ -1,8 +1,8 @@
1
1
  """How an env is served: the worker pool, where it binds, and each worker's bound.
2
2
 
3
3
  Serving is its own axis. `[env]` says what runs; this says how it's hosted, so an
4
- eval, a `serve` process, and a trainer configure it identically instead of each
5
- flattening pool knobs onto its own config."""
4
+ eval and a trainer's env server configure it identically instead of each flattening
5
+ pool knobs onto its own config."""
6
6
 
7
7
  from typing import Annotated, Literal
8
8
 
@@ -35,10 +35,10 @@ PoolConfig = Annotated[
35
35
  ]
36
36
 
37
37
 
38
- class ServingConfig(BaseConfig):
38
+ class ServeConfig(BaseConfig):
39
39
  """The `[serve]` block: the worker pool, the ZMQ address, and each worker's
40
- episode bound. Read by whoever hosts the env — the `serve` CLI, a server-backed
41
- eval, a trainer's orchestrator."""
40
+ episode bound. Read by whoever hosts the env — a server-backed eval, a
41
+ trainer's env-server process."""
42
42
 
43
43
  pool: PoolConfig = Field(default_factory=ElasticPoolConfig)
44
44
  """Worker-pool sizing. `elastic` (default) starts at one worker and scales up on
@@ -1,3 +1,5 @@
1
+ from collections.abc import AsyncIterator
2
+ from contextlib import asynccontextmanager
1
3
  from typing import Annotated
2
4
 
3
5
  from pydantic import Field
@@ -45,6 +47,22 @@ def make_runtime(config: RuntimeConfig, name: str | None = None) -> Runtime:
45
47
  return runtime
46
48
 
47
49
 
50
+ @asynccontextmanager
51
+ async def provision_runtime(
52
+ config: RuntimeConfig, name: str | None = None
53
+ ) -> AsyncIterator[Runtime]:
54
+ """Provision a box from `config` and tear it down on exit.
55
+
56
+ `start()` sits inside the `try`: a failed start may already hold a paid sandbox, so
57
+ it has to reach `stop()` (which is safe on a partially-started runtime)."""
58
+ runtime = make_runtime(config, name)
59
+ try:
60
+ await runtime.start()
61
+ yield runtime
62
+ finally:
63
+ await runtime.stop()
64
+
65
+
48
66
  def runtime_is_local(config: RuntimeConfig) -> bool:
49
67
  """Whether a runtime of this config exchanges host-local URLs without a public
50
68
  tunnel, read off the runtime class without provisioning one."""
@@ -71,5 +89,6 @@ __all__ = [
71
89
  "SubprocessRuntime",
72
90
  "SubprocessRuntimeInfo",
73
91
  "make_runtime",
92
+ "provision_runtime",
74
93
  "runtime_is_local",
75
94
  ]
@@ -2,6 +2,7 @@
2
2
 
3
3
  import asyncio
4
4
  import atexit
5
+ import base64
5
6
  import contextlib
6
7
  import hashlib
7
8
  import logging
@@ -285,9 +286,42 @@ class Runtime(ABC):
285
286
  argv = await self.prepare_uv_script(script, env)
286
287
  return await self.run([*argv, *(args or [])], env or {})
287
288
 
289
+ async def read(self, path: str, max_bytes: int | None = None) -> bytes:
290
+ """Read `path` into host memory. `max_bytes` caps the transfer, raising past
291
+ the cap — for a file written by something we don't control, whose size we
292
+ can't assume. The cap is enforced inside the box rather than after the
293
+ transfer, and base64 because `run` returns decoded text. Framework method —
294
+ override `_read`, not this."""
295
+ if max_bytes is None:
296
+ return await self._read(path)
297
+ # Through a temp file, not a pipe: `head | base64` exits with base64's 0
298
+ # even when the path is missing, and a missing file must raise here just
299
+ # as it does from `_read`.
300
+ result = await self.run(
301
+ [
302
+ "sh",
303
+ "-c",
304
+ (
305
+ "t=$(mktemp) || exit 1; "
306
+ 'head -c "$1" -- "$2" > "$t" || { rm -f "$t"; exit 1; }; '
307
+ 'base64 < "$t"; rc=$?; rm -f "$t"; exit $rc'
308
+ ),
309
+ "sh",
310
+ str(max_bytes + 1),
311
+ path,
312
+ ],
313
+ {},
314
+ )
315
+ if result.exit_code:
316
+ raise SandboxError(f"read {path!r}: {result.stderr.strip()[-500:]}")
317
+ data = base64.b64decode(result.stdout)
318
+ if len(data) > max_bytes:
319
+ raise SandboxError(f"read {path!r}: over the {max_bytes} byte limit")
320
+ return data
321
+
288
322
  @abstractmethod
289
- async def read(self, path: str) -> bytes:
290
- pass
323
+ async def _read(self, path: str) -> bytes:
324
+ """Read the whole file at `path`; `read` adds the optional transfer cap."""
291
325
 
292
326
  @abstractmethod
293
327
  async def write(self, path: str, data: bytes) -> None:
@@ -334,7 +334,7 @@ class DockerRuntime(Runtime):
334
334
  if run.exit_code != 0:
335
335
  raise SandboxError(f"docker exec -d failed: {run.stderr.strip()}")
336
336
 
337
- async def read(self, path: str) -> bytes:
337
+ async def _read(self, path: str) -> bytes:
338
338
  proc = await asyncio.create_subprocess_exec(
339
339
  "docker",
340
340
  "exec",
@@ -165,7 +165,7 @@ class ModalRuntime(Runtime):
165
165
  return path
166
166
  return f"{self.config.workdir.rstrip('/')}/{path}"
167
167
 
168
- async def read(self, path: str) -> bytes:
168
+ async def _read(self, path: str) -> bytes:
169
169
  try:
170
170
  return await self._sandbox.filesystem.read_bytes.aio(self._abs(path))
171
171
  except Exception as e:
@@ -270,7 +270,7 @@ class PrimeRuntime(Runtime):
270
270
  f"prime background launch failed: {result.stderr.strip()}"
271
271
  )
272
272
 
273
- async def read(self, path: str) -> bytes:
273
+ async def _read(self, path: str) -> bytes:
274
274
  # Avoid background-job log limits and base64 overhead by downloading binary data directly.
275
275
  # The temporary file is removed on every exit, and its byte read stays off the event loop.
276
276
  target = (
@@ -96,7 +96,7 @@ class SubprocessRuntime(Runtime):
96
96
  proc
97
97
  ) # killed in stop() — a host process won't die on its own
98
98
 
99
- async def read(self, path: str) -> bytes:
99
+ async def _read(self, path: str) -> bytes:
100
100
  return await asyncio.to_thread((self.workdir / path).read_bytes)
101
101
 
102
102
  async def write(self, path: str, data: bytes) -> None:
@@ -1,3 +1,4 @@
1
+ from verifiers.v1.tasksets.harbor.env import HarborEnv, HarborEnvConfig
1
2
  from verifiers.v1.tasksets.harbor.taskset import (
2
3
  HarborConfig,
3
4
  HarborData,
@@ -5,4 +6,11 @@ from verifiers.v1.tasksets.harbor.taskset import (
5
6
  HarborTaskset,
6
7
  )
7
8
 
8
- __all__ = ["HarborConfig", "HarborData", "HarborTask", "HarborTaskset"]
9
+ __all__ = [
10
+ "HarborConfig",
11
+ "HarborData",
12
+ "HarborEnv",
13
+ "HarborEnvConfig",
14
+ "HarborTask",
15
+ "HarborTaskset",
16
+ ]
@@ -0,0 +1,124 @@
1
+ """The harbor taskset's own env: the single solver seat, plus separate-verifier
2
+ grading for tasks that declare ``[verifier].environment_mode = "separate"``.
3
+
4
+ The default env for harbor runs (the taskset package exports it). A shared-verifier
5
+ task runs exactly as under the single-agent env: one `agent` trace, graded in the
6
+ box it worked in. A separate-verifier task is graded by `finalize` instead: the
7
+ solver's declared artifacts travel (collected by its task `finalize` while its box
8
+ is alive), a fresh box is provisioned from the task's verifier declaration,
9
+ `tests/` is staged there, and the verifier's rewards land on the solver's trace.
10
+ No second agent is involved — the verifier is the task's own `tests/test.sh`.
11
+ """
12
+
13
+ import asyncio
14
+ import logging
15
+ from contextlib import AsyncExitStack
16
+
17
+ from pydantic import Field
18
+
19
+ import verifiers.v1 as vf
20
+ from verifiers.v1.runtimes import RuntimeConfig, provision_runtime
21
+ from verifiers.v1.tasksets.harbor.taskset import (
22
+ HarborTask,
23
+ verifier_box_data,
24
+ )
25
+ from verifiers.v1.utils.artifacts import restore
26
+ from verifiers.v1.utils.compile import resolve_runtime_config
27
+ from verifiers.v1.utils.retries import backoff
28
+
29
+ logger = logging.getLogger(__name__)
30
+
31
+
32
+ class HarborEnvConfig(vf.EnvConfig):
33
+ agent: vf.AgentConfig = vf.AgentConfig()
34
+ """The one seat — the policy under evaluation/training; pin
35
+ `--env.agent.harness.*` to choose its program or runtime."""
36
+ verifier_runtime: RuntimeConfig | None = None
37
+ """Where a separate-verifier task grades. None derives the grading box from
38
+ the solver's runtime policy; set it (e.g. `--env.verifier-runtime.type prime
39
+ --env.verifier-runtime.vm true`) when the verifier needs different placement
40
+ than the agent."""
41
+ verifier_retries: int = Field(2, ge=0)
42
+ """Extra attempts at provisioning-and-grading the separate box before the
43
+ episode fails. Grading is deterministic; what these retry is the
44
+ infrastructure around it (image pulls, provisioning)."""
45
+
46
+
47
+ class HarborEnv(vf.Env[HarborEnvConfig]):
48
+ async def run(self, task: vf.Task, agents: vf.Agents) -> None:
49
+ if not isinstance(task, HarborTask):
50
+ raise TypeError(
51
+ f"the harbor env runs harbor tasks; got {type(task).__name__}"
52
+ )
53
+ if task.data.verifier is None:
54
+ await agents.agent.run(task)
55
+ return
56
+ # Resolve the verifier's box before the solve, so an impossible pairing
57
+ # (e.g. a restricted Prime verifier without vm=true) costs nothing
58
+ # rather than a full agent run.
59
+ self._verifier_config(task)
60
+ await agents.agent.run(task.graded_elsewhere())
61
+
62
+ def _verifier_config(self, task: HarborTask) -> RuntimeConfig:
63
+ base = (
64
+ self.config.verifier_runtime
65
+ if self.config.verifier_runtime is not None
66
+ else self.config.agent.runtime
67
+ )
68
+ return resolve_runtime_config(base, HarborTask(verifier_box_data(task.data)))
69
+
70
+ async def finalize(self, task: vf.Task, episode: vf.Episode) -> None:
71
+ """Grade a separate-verifier task in its own box, onto the solver's trace.
72
+
73
+ Provision a fresh box from the task's verifier declaration, restore the
74
+ solver's collected artifacts, stage `tests/`, run the verifier, and record
75
+ its rewards (and any extra reward.json keys as metrics) on the solver's
76
+ trace. Infrastructure failures retry per `verifier_retries`; the last one
77
+ fails the episode — a grading box that can't be reached must never read
78
+ as reward 0."""
79
+ if not isinstance(task, HarborTask) or task.data.verifier is None:
80
+ return
81
+ solution = episode.traces[0]
82
+ if not solution.ok:
83
+ return
84
+ grader = HarborTask(verifier_box_data(task.data))
85
+ scores = await self._grade(self._verifier_config(task), grader, solution)
86
+ items = scores.items() if isinstance(scores, dict) else [("solved", scores)]
87
+ for name, value in items:
88
+ solution.record_reward(name, value)
89
+
90
+ async def _grade(
91
+ self, config: RuntimeConfig, grader: HarborTask, solution: vf.Trace
92
+ ) -> float | dict[str, float]:
93
+ last: Exception | None = None
94
+ for attempt in range(self.config.verifier_retries + 1):
95
+ if attempt:
96
+ delay = backoff(attempt - 1)
97
+ logger.warning(
98
+ "harbor verifier attempt %d/%d failed (%s); retrying in %.1fs",
99
+ attempt,
100
+ self.config.verifier_retries + 1,
101
+ last,
102
+ delay,
103
+ )
104
+ await asyncio.sleep(delay)
105
+ try:
106
+ # The scoring deadline covers provisioning and grading, but not the
107
+ # box's teardown: a score already in hand must not be discarded
108
+ # because the teardown ran out the clock.
109
+ async with AsyncExitStack() as boxes:
110
+ async with asyncio.timeout(grader.data.timeout.scoring):
111
+ box = await boxes.enter_async_context(provision_runtime(config))
112
+ await box.prepare_setup()
113
+ # Artifacts first, tests second: an artifact entry pointing
114
+ # into /tests must not survive staging, which wipes and
115
+ # rebuilds that directory.
116
+ await restore(box, solution.state.artifacts)
117
+ await grader._stage_tests(box, wipe=True)
118
+ await box.prepare_execution([])
119
+ scores = await grader._graded(box, solution)
120
+ return scores
121
+ except Exception as e: # noqa: BLE001 - each attempt's failure is retried
122
+ last = e
123
+ assert last is not None
124
+ raise last
@@ -1,18 +1,25 @@
1
1
  """Harbor tasksets backed by Harbor Hub packages.
2
2
 
3
3
  The Harbor CLI downloads and caches each task directory. Its verifier runs in the
4
- same runtime the harness edited, then writes the score to
4
+ runtime the harness edited — or, when the task asks for it with
5
+ ``[verifier].environment_mode = "separate"``, in a second box the agent never
6
+ touched, carrying only what the task declared — the harbor env provisions and
7
+ grades that box (see ``env.py``). Either way the score lands in
5
8
  ``/logs/verifier/reward.json`` or the legacy ``reward.txt``.
6
9
 
7
10
  A pullable ``[environment].docker_image`` becomes ``TaskData.image``. Verifiers does
8
11
  not build Dockerfile-only environments, so those are rejected unless ``ignore_dockerfile``
9
12
  deliberately uses the harness runtime image. Tasks without an environment also use that
10
- image unless ``require_image`` is set.
13
+ image unless ``require_image`` is set. The same rule applies to a declared
14
+ ``[verifier.environment]``: it needs a pullable ``docker_image``, since Harbor would
15
+ otherwise build the verifier image from ``tests/Dockerfile``.
11
16
  """
12
17
 
13
18
  import asyncio
19
+ import copy
14
20
  import hashlib
15
21
  import io
22
+ import logging
16
23
  import shutil
17
24
  import subprocess
18
25
  import sys
@@ -26,7 +33,7 @@ from typing import Annotated
26
33
  from pydantic import BaseModel, ConfigDict, Field, TypeAdapter, ValidationError
27
34
 
28
35
  from verifiers.v1.configs.taskset import TasksetConfig
29
- from verifiers.v1.errors import SandboxError
36
+ from verifiers.v1.errors import SandboxError, TaskError
30
37
  from verifiers.v1.runtimes import Runtime
31
38
  from verifiers.v1.task import Task, TaskData, TaskResources, TaskTimeout
32
39
  from verifiers.v1.taskset import Taskset
@@ -34,9 +41,12 @@ from verifiers.v1.trace import Trace
34
41
  from verifiers.v1.utils.artifacts import Artifact, collect
35
42
  from verifiers.v1.utils.decorators import reward
36
43
 
44
+ logger = logging.getLogger(__name__)
45
+
37
46
  CACHE = Path.home() / ".cache" / "harbor"
38
47
  HARBOR_INSTALL_HINT = "uv sync --python 3.12 --extra harbor"
39
48
  REWARD_JSON = "/logs/verifier/reward.json"
49
+ MAX_REWARD_BYTES = 1024 * 1024
40
50
  REWARD_JSON_ADAPTER = TypeAdapter(
41
51
  float | Annotated[dict[str, float], Field(min_length=1)],
42
52
  config=ConfigDict(strict=True, allow_inf_nan=False),
@@ -76,6 +86,11 @@ class HarborConfig(TasksetConfig):
76
86
  instead of rejecting it. The Dockerfile is NOT built, so the task scores against the
77
87
  harness image rather than its declared environment — only correct when that image already
78
88
  has what the task needs (e.g. you've pointed the runtime at the right image)."""
89
+ ignore_separate_verifier: bool = False
90
+ """Grade every task in the agent's own box, even one whose `[verifier]` asks for a
91
+ separate one. Trades the isolation for a sandbox per task; useful when provisioning
92
+ is the bottleneck. Note what it gives up: the grader becomes reachable by the agent
93
+ that just ran."""
79
94
 
80
95
 
81
96
  class Author(BaseModel):
@@ -90,6 +105,27 @@ class CollectHook(BaseModel):
90
105
  timeout_sec: float = 600.0
91
106
 
92
107
 
108
+ class VerifierConfig(BaseModel):
109
+ """The box this task's verifier wants, when it wants one of its own.
110
+
111
+ `None` on `HarborData` means shared — grade where the agent worked, which is still
112
+ Harbor's default and every task that says nothing."""
113
+
114
+ image: str | None = None
115
+ """Pullable ref from `[verifier.environment].docker_image`. None keeps the task's
116
+ own image, which is what Harbor's fresh copy of `[environment]` resolves to."""
117
+ resources: TaskResources = TaskResources()
118
+ workdir: str | None = None
119
+ fresh_copy: bool = False
120
+ """Whether this came from Harbor's fresh copy of `[environment]` rather than a
121
+ declared `[verifier.environment]`. A fresh copy inherits the agent box's resolved
122
+ resources; a declared environment states its own, and what it omits falls back to
123
+ the run's rather than to the agent's task-derived values."""
124
+ network_allow: list[str] = Field(default_factory=lambda: ["*"])
125
+ """Destinations the verifier may reach, from the verifier's network mode. `["*"]`
126
+ is unrestricted; `[]` is Harbor's `no-network` / `allow_internet = false`."""
127
+
128
+
93
129
  class HarborData(TaskData):
94
130
  """Parsed ``task.toml`` metadata plus the host-side verifier directory.
95
131
 
@@ -111,6 +147,9 @@ class HarborData(TaskData):
111
147
  collect: list[CollectHook] = Field(default_factory=list)
112
148
  """`[[verifier.collect]]` blocks: commands that snapshot runtime state into files
113
149
  after the agent stops, so the files can travel to a grading box as artifacts."""
150
+ verifier: VerifierConfig | None = None
151
+ """The verifier's own box, when `[verifier].environment_mode` asks for one. None
152
+ grades in the agent's box."""
114
153
 
115
154
 
116
155
  class HarborTask(Task[HarborData]):
@@ -145,30 +184,59 @@ class HarborTask(Task[HarborData]):
145
184
  )
146
185
  trace.state.artifacts = await collect(runtime, self.data.artifacts)
147
186
 
148
- @reward(weight=1.0)
149
- async def solved(self, runtime: Runtime, trace: Trace) -> float | dict[str, float]:
187
+ def graded_elsewhere(self) -> "HarborTask":
188
+ """A copy whose `solved` records nothing here: the harbor env grades this
189
+ task's finished work in a separate box of the task's choosing."""
190
+ clone = copy.copy(self)
191
+ clone._graded_elsewhere = True
192
+ return clone
193
+
194
+ _graded_elsewhere: bool = False
195
+
196
+ async def _stage_tests(self, runtime: Runtime, wipe: bool = False) -> None:
197
+ """Put the task package's `tests/` in `/tests`, where `test.sh` expects it.
198
+
199
+ Raises rather than scoring stale state: a leftover reward file — planted by
200
+ the agent or shipped in the image — must be gone before `test.sh` runs, so a
201
+ removal that fails must not fall through to reading it.
202
+
203
+ `wipe` for a box we did not watch being built: a fresh container of the task's
204
+ image can ship its own `/tests`, and a leftover file there would be graded as
205
+ though it came from the package.
206
+ """
150
207
  await runtime.write(
151
208
  "/tmp/tests.tgz", make_tar(Path(self.data.task_dir) / "tests")
152
209
  )
153
- await runtime.run(
154
- [
155
- "sh",
156
- "-c",
157
- "mkdir -p /logs/verifier /tests && tar -xzf /tmp/tests.tgz -C /tests",
158
- ],
159
- {},
160
- )
161
- await runtime.run(
162
- [
163
- "sh",
164
- "-c",
165
- (
166
- "rm -f /logs/verifier/reward.json /logs/verifier/reward.txt"
167
- " && cd /tests && bash test.sh"
168
- ),
169
- ],
170
- verifier_env(self.data),
210
+ stage = (
211
+ f"{'rm -rf /tests && ' if wipe else ''}"
212
+ "rm -f /logs/verifier/reward.json /logs/verifier/reward.txt && "
213
+ "mkdir -p /logs/verifier /tests && tar -xzf /tmp/tests.tgz -C /tests"
171
214
  )
215
+ result = await runtime.run(["sh", "-c", stage], {})
216
+ if result.exit_code:
217
+ raise TaskError(
218
+ f"staging tests failed (exit {result.exit_code}): "
219
+ f"{(result.stderr or result.stdout).strip()[-500:]}"
220
+ )
221
+
222
+ @reward(weight=1.0)
223
+ async def solved(self, runtime: Runtime, trace: Trace) -> float | dict[str, float]:
224
+ if self.data.verifier is not None:
225
+ if not self._graded_elsewhere:
226
+ raise TaskError(
227
+ f"task {self.data.name!r} declares a separate verifier "
228
+ '([verifier].environment_mode = "separate"); grade it through '
229
+ "the harbor env (this taskset's default), or force shared "
230
+ "grading with --taskset.ignore-separate-verifier"
231
+ )
232
+ return {}
233
+ await self._stage_tests(runtime)
234
+ return await self._graded(runtime, trace)
235
+
236
+ async def _graded(self, runtime: Runtime, trace: Trace) -> float | dict[str, float]:
237
+ # By absolute path, in the runtime's configured workdir: Harbor execs the
238
+ # script the same way, and scripts do grade the agent's work at `$PWD`.
239
+ await runtime.run(["bash", "/tests/test.sh"], verifier_env(self.data))
172
240
  scores = await self._reward_json(runtime)
173
241
  if scores is not None:
174
242
  if isinstance(scores, dict) and "reward" in scores:
@@ -178,19 +246,75 @@ class HarborTask(Task[HarborData]):
178
246
  return {"reward": scores["reward"]}
179
247
  return scores
180
248
  try:
181
- reward = (await runtime.read("/logs/verifier/reward.txt")).decode().strip()
249
+ reward = (
250
+ (
251
+ await runtime.read(
252
+ "/logs/verifier/reward.txt", max_bytes=MAX_REWARD_BYTES
253
+ )
254
+ )
255
+ .decode()
256
+ .strip()
257
+ )
182
258
  return float(reward or 0)
183
259
  except (SandboxError, OSError, ValueError):
184
260
  return 0.0
185
261
 
186
262
  async def _reward_json(self, runtime: Runtime) -> float | dict[str, float] | None:
187
- """Read Harbor's scalar or keyed JSON reward, if it is valid."""
263
+ """Read Harbor's scalar or keyed JSON reward, if it is valid.
264
+
265
+ Bounded: this is a grading input, and nothing guarantees its size.
266
+ """
188
267
  try:
189
- return REWARD_JSON_ADAPTER.validate_json(await runtime.read(REWARD_JSON))
268
+ return REWARD_JSON_ADAPTER.validate_json(
269
+ await runtime.read(REWARD_JSON, max_bytes=MAX_REWARD_BYTES)
270
+ )
190
271
  except (SandboxError, OSError, ValidationError):
191
272
  return None
192
273
 
193
274
 
275
+ def verifier_box_data(data: HarborData) -> HarborData:
276
+ """The verifier's box, declared as task data — the harbor env resolves the
277
+ grading runtime from it (image, workdir, resources, network policy), exactly
278
+ as the solver's box resolves from the solver task's.
279
+
280
+ Which box follows Harbor: a declared `[verifier.environment]` states its own
281
+ image, workdir, and resources, and what it omits is the run's default; a
282
+ fresh copy of `[environment]` keeps the task's own. The verifier's network
283
+ policy applies either way."""
284
+ verifier = data.verifier
285
+ if verifier is None:
286
+ raise TaskError(f"task {data.name!r} declares no separate verifier")
287
+ fresh = verifier.fresh_copy
288
+ return data.model_copy(
289
+ update={
290
+ "name": f"{data.name} (verifier)",
291
+ "image": verifier.image if verifier.image is not None else data.image,
292
+ "workdir": data.workdir if fresh else verifier.workdir,
293
+ "resources": data.resources if fresh else verifier.resources,
294
+ "network_allow": list(verifier.network_allow),
295
+ "network_block": [],
296
+ }
297
+ )
298
+
299
+
300
+ def task_resources(environment, multiplier: float) -> TaskResources:
301
+ """Harbor environment resource requests, scaled, as `TaskResources`.
302
+
303
+ Harbor declares CPU counts and MB sizes; `TaskResources` wants counts and GB.
304
+ GPU requests are never scaled.
305
+ """
306
+ return TaskResources(
307
+ cpu=environment.cpus * multiplier if environment.cpus else None,
308
+ memory=environment.memory_mb / 1024 * multiplier
309
+ if environment.memory_mb
310
+ else None,
311
+ gpu=str(environment.gpus) if environment.gpus else None,
312
+ disk=environment.storage_mb / 1024 * multiplier
313
+ if environment.storage_mb
314
+ else None,
315
+ )
316
+
317
+
194
318
  def harbor_cli() -> str:
195
319
  scripts_dir = Path(sys.executable).parent
196
320
  harbor_bin = shutil.which("harbor", path=str(scripts_dir))
@@ -318,7 +442,7 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
318
442
 
319
443
  harbor_task = HarborModelTask(task_dir)
320
444
  parsed = harbor_task.config
321
- artifacts, collect = parse_verifier_extras(task_dir, parsed)
445
+ artifacts, hooks, verifier = parse_verifier_extras(task_dir, parsed, harbor_config)
322
446
  environment = parsed.environment
323
447
  network = parsed.agent.explicit_phase_policy() or environment.resolve_baseline()
324
448
  task, meta = parsed.task, parsed.metadata
@@ -368,18 +492,7 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
368
492
  if scoring_timeout is not None
369
493
  else None,
370
494
  ),
371
- resources=TaskResources(
372
- cpu=environment.cpus * harbor_config.resource_multiplier
373
- if environment.cpus
374
- else None,
375
- memory=environment.memory_mb / 1024 * harbor_config.resource_multiplier
376
- if environment.memory_mb
377
- else None,
378
- gpu=str(environment.gpus) if environment.gpus else None,
379
- disk=environment.storage_mb / 1024 * harbor_config.resource_multiplier
380
- if environment.storage_mb
381
- else None,
382
- ),
495
+ resources=task_resources(environment, harbor_config.resource_multiplier),
383
496
  keywords=task.keywords if task else [],
384
497
  authors=authors,
385
498
  difficulty=meta.get("difficulty"),
@@ -388,14 +501,22 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
388
501
  task_dir=str(task_dir),
389
502
  verifier_env=parsed.verifier.env,
390
503
  artifacts=artifacts,
391
- collect=collect,
504
+ collect=hooks,
505
+ verifier=verifier,
392
506
  )
393
507
 
394
508
 
395
509
  def parse_verifier_extras(
396
- task_dir: Path, parsed
397
- ) -> tuple[list[Artifact], list[CollectHook]]:
398
- """Parse supported artifact and collect-hook settings."""
510
+ task_dir: Path, parsed, harbor_config: HarborConfig
511
+ ) -> tuple[list[Artifact], list[CollectHook], VerifierConfig | None]:
512
+ """Harbor's `artifacts`, `[[verifier.collect]]` blocks, and verifier environment,
513
+ narrowed to what verifiers' verifier-runtime integration can honor.
514
+
515
+ The convention dir is deliberately not prepended here (Harbor's
516
+ `with_convention_entry` would): collection injects it itself, as an optional sweep.
517
+ Prepending it would make it an explicitly declared entry, and declared entries are
518
+ required — which would fail every task that never writes there.
519
+ """
399
520
  from harbor.constants import MAIN_SERVICE_NAME
400
521
  from harbor.models.task.artifacts import (
401
522
  effective_artifact_service,
@@ -403,13 +524,6 @@ def parse_verifier_extras(
403
524
  )
404
525
 
405
526
  verifier = parsed.verifier
406
- if verifier.environment is not None:
407
- raise ValueError(
408
- f"{task_dir.name}: [verifier.environment] declares a separate verifier "
409
- "image. Grading runs in a fresh box built from the task's own image, so "
410
- "only the agent's delta has to travel; a different verifier image needs "
411
- "the full working tree copied over and isn't supported yet."
412
- )
413
527
  if verifier.user is not None:
414
528
  raise ValueError(f"{task_dir.name}: [verifier].user is not supported")
415
529
 
@@ -447,7 +561,89 @@ def parse_verifier_extras(
447
561
  )
448
562
  hooks.append(CollectHook(command=hook.command, timeout_sec=hook.timeout_sec))
449
563
 
450
- return artifacts, hooks
564
+ return artifacts, hooks, parse_verifier_environment(task_dir, parsed, harbor_config)
565
+
566
+
567
+ def parse_verifier_environment(
568
+ task_dir: Path, parsed, harbor_config: HarborConfig
569
+ ) -> VerifierConfig | None:
570
+ """The box Harbor wants this task's verifier in, or None to grade in the agent's.
571
+
572
+ Harbor resolves `[verifier.environment]` if declared, else a deep copy of
573
+ `[environment]` — so a mode-only `separate` lands on the task's own image and needs
574
+ nothing but a second box. A declared environment is the case that can name a
575
+ different image, and the case that can name none at all: there Harbor builds
576
+ `tests/Dockerfile`, which verifiers never does.
577
+ """
578
+ from harbor.models.task.config import NetworkMode, TaskOS
579
+ from harbor.models.task.verifier_mode import (
580
+ VerifierEnvironmentMode,
581
+ resolve_effective_verifier_env_config,
582
+ resolve_task_verifier_mode,
583
+ )
584
+
585
+ if resolve_task_verifier_mode(parsed) != VerifierEnvironmentMode.SEPARATE:
586
+ return None
587
+ if harbor_config.ignore_separate_verifier:
588
+ logger.warning(
589
+ "%s: asks for a separate verifier; grading in the agent's box anyway "
590
+ "(--taskset.ignore-separate-verifier)",
591
+ task_dir.name,
592
+ )
593
+ return None
594
+
595
+ environment = resolve_effective_verifier_env_config(parsed, None)
596
+ if environment is None: # unreachable while the mode is SEPARATE
597
+ raise ValueError(f"{task_dir.name}: separate verifier resolved no environment")
598
+ declared = parsed.verifier.environment is not None
599
+
600
+ if declared and environment.docker_image is None:
601
+ if not harbor_config.ignore_dockerfile:
602
+ raise ValueError(
603
+ f"{task_dir.name}: [verifier.environment] names no docker_image, so "
604
+ "Harbor would build the verifier image from tests/Dockerfile. Verifiers "
605
+ "pulls images and never builds them: build and push it yourself (e.g. "
606
+ "`prime images push`) and set [verifier.environment].docker_image to the "
607
+ "resulting ref, or pass --taskset.ignore-dockerfile to grade in the "
608
+ "agent's image instead."
609
+ )
610
+ logger.warning(
611
+ "%s: [verifier.environment] names no docker_image — grading in the agent's "
612
+ "image rather than building tests/Dockerfile, so the verifier runs somewhere "
613
+ "the task never declared",
614
+ task_dir.name,
615
+ )
616
+ unsupported = [
617
+ field
618
+ for field in ("healthcheck", "mcp_servers", "skills_dir", "gpu_types", "tpu")
619
+ if getattr(environment, field, None)
620
+ ]
621
+ if environment.os != TaskOS.LINUX or unsupported:
622
+ raise ValueError(
623
+ f"{task_dir.name}: verifier environment declares "
624
+ f"{unsupported or environment.os}, which verifiers' verifier-runtime "
625
+ "integration cannot honor"
626
+ )
627
+
628
+ network = parsed.verifier.explicit_phase_policy() or environment.resolve_baseline()
629
+ return VerifierConfig(
630
+ image=environment.docker_image if declared else None,
631
+ # A declared environment states its own resources; what it leaves out is the
632
+ # run's default, not the agent task's. A fresh copy is the task's environment,
633
+ # so it keeps whatever the agent box resolved to.
634
+ resources=(
635
+ task_resources(environment, harbor_config.resource_multiplier)
636
+ if declared
637
+ else TaskResources()
638
+ ),
639
+ workdir=environment.workdir if declared else None,
640
+ fresh_copy=not declared,
641
+ network_allow=(
642
+ ["*"]
643
+ if network.network_mode == NetworkMode.PUBLIC
644
+ else list(network.allowed_hosts)
645
+ ),
646
+ )
451
647
 
452
648
 
453
649
  def verifier_env(task: HarborData) -> dict[str, str]:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: verifiers
3
- Version: 0.2.2.dev74
3
+ Version: 0.2.2.dev76
4
4
  Summary: Verifiers: Environments for LLM Reinforcement Learning
5
5
  Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
6
6
  Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
@@ -167,8 +167,8 @@ verifiers/utils/threaded_sandbox_client.py,sha256=XRzYhWBpMev7kw_aNMWAy6lfa-9uAn
167
167
  verifiers/utils/tool_utils.py,sha256=gInWZODWQUZN4TyEMuuITyOUm9qlHJxtcvr1zZl_zJs,1020
168
168
  verifiers/utils/usage_utils.py,sha256=Mw6WC78r6EsLyRFdP9ICveqptkjizbSkAC86ZH0geVU,3918
169
169
  verifiers/utils/version_utils.py,sha256=39gV8-mNF9GspAgKloLo8oy7_MVluMrwGPrntQv2hx4,2656
170
- verifiers/v1/__init__.py,sha256=kacTRsGsRHfOruv16ZyIOesfvCH_nW98ARKPYSgcNIc,7509
171
- verifiers/v1/agent.py,sha256=3kQNASeO0aedCN8m95RAzxu2TbK0rfG7Ix6-f2yxQ7Q,31545
170
+ verifiers/v1/__init__.py,sha256=v6qrJQFeNL2b4wd0PsDWiituw2Y-f5rMaFmdbdP6SdM,7505
171
+ verifiers/v1/agent.py,sha256=MQVKGjUhrP8uzRXAq94uJHkEsdHFziPxQHZ7TOD324I,31306
172
172
  verifiers/v1/env.py,sha256=g6fpoG-Z9lM3vTKrrU-QRJY3S88_y9zsPHwBvfs5vdY,17865
173
173
  verifiers/v1/episode.py,sha256=YtuRTV_PpwDglXFeN6UNg8IswpP1HseFLCYtascGMNI,3218
174
174
  verifiers/v1/errors.py,sha256=kQeEPX06TwAIuiz7MlWFYirrpKbx5STi6A5b_dtQQoA,6885
@@ -192,7 +192,6 @@ verifiers/v1/cli/init.py,sha256=Qj96Kwzfhaejhrc0yYqHNRl3ncoID028RGlHvS1C98Q,8508
192
192
  verifiers/v1/cli/output.py,sha256=pMS5Z0EBgyAVShNCAk9jVURNeLOaTfeRWpt0J8hHjq8,5890
193
193
  verifiers/v1/cli/replay.py,sha256=DAh9kB5PB-ppZLGhxdJwrCPch_Yky5HKhplwGFhKEz0,10164
194
194
  verifiers/v1/cli/resolve.py,sha256=OsNw4C3r9xOVFa-mminJFcex_bAsJ6EQ8K7A_H8u2kQ,4370
195
- verifiers/v1/cli/serve.py,sha256=_mdJaHVVe4pXzcpAot0ugcmiN3ra2gS4GrGmLghnX5M,2866
196
195
  verifiers/v1/cli/validate.py,sha256=7ax6FqIzNBSYjd3JXK4v7Vbq34Rozyh1OvGpYnGqlSg,10266
197
196
  verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
198
197
  verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
@@ -216,16 +215,15 @@ verifiers/v1/configs/harness.py,sha256=CG8hGbHl3rr6yawH3YYMAFl01M7Piiwf4w5oiw2Va
216
215
  verifiers/v1/configs/judge.py,sha256=O7oe28tS24fKRXUML_-6vHrVGdG6w7061SPQXOH7XvM,2223
217
216
  verifiers/v1/configs/legacy.py,sha256=YOJM6L5u6-jYpZ1CLpPInufKdmz1lEsMNKIRdT2JaaY,1080
218
217
  verifiers/v1/configs/retries.py,sha256=-nPlgmz7J_NZ4GOYG6oCc3thXo4La2SufBmvQGvvPGo,821
219
- verifiers/v1/configs/serve.py,sha256=cHHnFGQtqEwvaovdeoVMpCebTHgPXKUuIivN5IhxZR0,2596
218
+ verifiers/v1/configs/serve.py,sha256=nSoJcTX2S0Vy9EUeNqW0Yo-f9_MUqpBiY-Q9FF_4cRo,2576
220
219
  verifiers/v1/configs/task.py,sha256=1ozZjfpahin7I_TqzC7f3xr7CHSthAdydegFPXW4d9k,1101
221
220
  verifiers/v1/configs/taskset.py,sha256=o-tr1xkmGmYG10_ySG-Du_djWS-dN4h7cv14WSvJ4eY,851
222
- verifiers/v1/configs/cli/__init__.py,sha256=3wBAONw5XkBPqbMjpHk5XBHlBVybR8JQ2JCPWijYBMU,444
221
+ verifiers/v1/configs/cli/__init__.py,sha256=TJaWQQHrcyXoeqS7P9S25ztd78YJ30_9_Ug0czRH7Tw,374
223
222
  verifiers/v1/configs/cli/debug.py,sha256=VFsZ1tiKGvvx5mFDOzqqeU4cRtQkJ8OeD8mmRBZhy8w,2849
224
223
  verifiers/v1/configs/cli/env.py,sha256=DJbJy0-DqqtmFm0r7NpvRBbuxXnXc1o-yurDKUBSGJk,2701
225
- verifiers/v1/configs/cli/eval.py,sha256=YWYMvPOIAG51e0JdmyA-U_P7eInMPZ_8DTFim6aWn6Y,6869
224
+ verifiers/v1/configs/cli/eval.py,sha256=HqIBtt8DuZEHZ07KgtE1jiqZy8OsDMfTahTuBAvslcA,6863
226
225
  verifiers/v1/configs/cli/init.py,sha256=tMaTDF_lU1Af3JmQpksk2mwe8CAfpnHsquXhwepZhYo,1019
227
226
  verifiers/v1/configs/cli/replay.py,sha256=xvMDZzIkvjgYl-a3ragjXI65DsD6FYl6gEFM5qpdIuA,2913
228
- verifiers/v1/configs/cli/serve.py,sha256=1rAt0QOL-XT3DMmRUIA40t1E3YuoD1vOADfsiTwFBf0,2981
229
227
  verifiers/v1/configs/cli/validate.py,sha256=UdgC6bQ2RYx3t8n9HjsImAHV_ifoOxYOSlbtdZzOnZY,2298
230
228
  verifiers/v1/dialects/__init__.py,sha256=p72C7VYrFQzVgTcfmzpHGSwXlvThfc4pvjBxsqw7kHE,786
231
229
  verifiers/v1/dialects/anthropic.py,sha256=nj9J7OOhexmPHa-sWiNP2xlI9_bGhvwM5VWzdh19pBc,13526
@@ -297,13 +295,13 @@ verifiers/v1/mcp/__init__.py,sha256=pm8ZWBiL1q1tY4wHvJzM_QrZ9oR8aGm-bIRWiRJsgHI,
297
295
  verifiers/v1/mcp/launch.py,sha256=wBpyeiOO86l1H2lVMbLVSV9Lbp90XeyXqauuK8snfEU,20171
298
296
  verifiers/v1/mcp/server.py,sha256=dRBxytFgS6rv1BS6t5EgBd3xeF6U1nTyt0ArDCLk3VE,12301
299
297
  verifiers/v1/mcp/toolset.py,sha256=T8Ioiq4wphxVOTHUDf28ZoQBM2ZYEevGnGgw8TFEMiA,1002
300
- verifiers/v1/runtimes/__init__.py,sha256=TqNtJVahzQuLS4fgCVeT9xdOzTkp0hRAPOkl7eFVDUg,2011
301
- verifiers/v1/runtimes/base.py,sha256=sTCiLo51yAU_T-Voc3BTH-FuzyEM0cGY05VzFE4DmYc,14720
298
+ verifiers/v1/runtimes/__init__.py,sha256=Kwe612RjtGzrQhc19QdXc7CmZAnQLH-EdK_xqkKP3Tk,2629
299
+ verifiers/v1/runtimes/base.py,sha256=tZRELSfZmi3l_1BsjabnNC3mEPfqNvUNevdnRvLPv6w,16257
302
300
  verifiers/v1/runtimes/limiters.py,sha256=ZjjYUFEt20yH8ggL5uU9w9U500AOB44qSUTQ7bYyh0c,2730
303
- verifiers/v1/runtimes/modal.py,sha256=FnocoJd9lQVQS3V3CoBon7KpKXm1bKWWeU2FAkq2bKU,8825
304
- verifiers/v1/runtimes/prime.py,sha256=zMMoB85uqwXjRTTyb3q5C7Xs4VmHagB8oI92Kuc-E1E,14755
305
- verifiers/v1/runtimes/subprocess.py,sha256=cs985D8-n_DOk88IRTyP1vfVlyUV1NBvS450lC-vw_s,5987
306
- verifiers/v1/runtimes/docker/__init__.py,sha256=eC5hjfyPwvwEKaeZUeP9EN8tJkrl0HUxGMzjiK4tK4Q,14555
301
+ verifiers/v1/runtimes/modal.py,sha256=eFiT-Ovf_RwQZbSw3Dwv2EXeHV0d6uGvPswqKYKCw5k,8826
302
+ verifiers/v1/runtimes/prime.py,sha256=kwvPqUI8D-GvcFwYRdpTHE1hXS0vfJGPlFOLgzm4Mvs,14756
303
+ verifiers/v1/runtimes/subprocess.py,sha256=ZrHRlTuTyCrcSKpiPD2ElcPpe97EvVsXJ8BfwUaNXiI,5988
304
+ verifiers/v1/runtimes/docker/__init__.py,sha256=vyEgf6XgAwjGgBicJchJvNgcDgYXrRvyQC4W52-23GU,14556
307
305
  verifiers/v1/runtimes/docker/egress.py,sha256=37UXgNX9gnoLxjMg4KZaJIGoApCAM6MK22eJGNoMi38,14514
308
306
  verifiers/v1/serve/__init__.py,sha256=sr_zkDzAHnOcQm-QRm8xwJBmCMFD-xhKIFPaDZvMbUs,641
309
307
  verifiers/v1/serve/client.py,sha256=CxpOwCn4ehuCXfqDJeV9zj7O0XC7MPcI-OZ1L_CSDlo,7057
@@ -311,8 +309,9 @@ verifiers/v1/serve/pool.py,sha256=AyOMjDYi9PLGA8Z-bnr6Fd5cnubuRavEBuudbVWtq2E,14
311
309
  verifiers/v1/serve/server.py,sha256=eFWJsTUDbMywKQXKDuk57QvS7Y4IZwW2Ok3LwJHMFLI,9200
312
310
  verifiers/v1/serve/types.py,sha256=3F4IIGIQgYITlEFxeX3uNJYrJlKugFrPTozbDNwDfBY,2674
313
311
  verifiers/v1/tasksets/__init__.py,sha256=b62O4WDNbGfJH_4NjYmznn5otWovdfvUgJZyoIZyLh8,596
314
- verifiers/v1/tasksets/harbor/__init__.py,sha256=wwVGA8N7E2Q2qA5WGHuLRrn4DgG5oaRHrCS2RW3NXVY,195
315
- verifiers/v1/tasksets/harbor/taskset.py,sha256=9rfWTg0s3zPAEppZnjh7zNyi2EbM-qORK2lyWQZQL7s,19565
312
+ verifiers/v1/tasksets/harbor/__init__.py,sha256=JxThzHFYEkuSfQbVf3pEnjordgdfe4MU-QOsSSbyovU,326
313
+ verifiers/v1/tasksets/harbor/env.py,sha256=bNF5dfbhcsdrd_8JmW5ved8Yk-mHhrOgKkVEA563qEo,5847
314
+ verifiers/v1/tasksets/harbor/taskset.py,sha256=vmDjP7dhxchZ5-J9IUeiATwwchxUvAeciCevD6W_N4c,29032
316
315
  verifiers/v1/tasksets/lean/__init__.py,sha256=oyv-GbJQBvCyGwo7Ysxm2NjZW-aX4j1uDoHU0f-E7xQ,767
317
316
  verifiers/v1/tasksets/lean/scoring.py,sha256=sfzsT6MUK0zdMgAQ57_P_0CBaPHy3M7go71QMr5hDrs,10599
318
317
  verifiers/v1/tasksets/lean/taskset.py,sha256=XgUqX0uJvHC8sKFhbK92iu6nUGcPuXCUpI3ZBhDcb7c,9025
@@ -338,8 +337,8 @@ verifiers/v1/utils/platform.py,sha256=56Ixmk1SER6q5LvzyYcA0hgzGpzhxKZL8Hiqp4c_XV
338
337
  verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
339
338
  verifiers/v1/utils/score.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
340
339
  verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
341
- verifiers-0.2.2.dev74.dist-info/METADATA,sha256=mYHTNMUxzf9tA02Dz787WOdGwU54MPGEJBgTSFG6aSs,4545
342
- verifiers-0.2.2.dev74.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
343
- verifiers-0.2.2.dev74.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
344
- verifiers-0.2.2.dev74.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
345
- verifiers-0.2.2.dev74.dist-info/RECORD,,
340
+ verifiers-0.2.2.dev76.dist-info/METADATA,sha256=vOCZL_bCXmOR9KHHvbzSCszxDmGtjDRTNxI0I8Zn8Hs,4545
341
+ verifiers-0.2.2.dev76.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
342
+ verifiers-0.2.2.dev76.dist-info/entry_points.txt,sha256=v6v0QT9vVExnfn4br42MostI_o3f3ZbWYlTormz-U3g,515
343
+ verifiers-0.2.2.dev76.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
344
+ verifiers-0.2.2.dev76.dist-info/RECORD,,
@@ -4,7 +4,6 @@ eval = verifiers.v1.cli.eval.main:main
4
4
  gepa = verifiers.v1.cli.gepa:main
5
5
  init = verifiers.v1.cli.init:main
6
6
  replay = verifiers.v1.cli.replay:main
7
- serve = verifiers.v1.cli.serve:main
8
7
  validate = verifiers.v1.cli.validate:main
9
8
  vf-build = verifiers.scripts.build:main
10
9
  vf-eval = verifiers.scripts.eval:main
verifiers/v1/cli/serve.py DELETED
@@ -1,79 +0,0 @@
1
- """Environment-server CLI entrypoint."""
2
-
3
- import sys
4
- from functools import partial
5
-
6
- from pydantic_config import cli
7
-
8
- from verifiers.v1.cli.resolve import (
9
- extract_id,
10
- narrow_config,
11
- plugin_errors,
12
- references_config_file,
13
- with_positional_taskset,
14
- )
15
- from verifiers.v1.configs.cli.serve import ServeConfig
16
- from verifiers.v1.configs.serve import pool_serve_kwargs
17
- from verifiers.v1.serve import serve_env
18
- from verifiers.v1.utils.logging import setup_logging
19
-
20
- USAGE = "usage: uv run serve [<taskset-id>] [--env.id <id>] [--legacy.id <env-id> (v0)] [options] [@ file.toml]"
21
-
22
-
23
- def main(argv: list[str] | None = None) -> None:
24
- argv = with_positional_taskset(list(sys.argv[1:]) if argv is None else list(argv))
25
-
26
- if not argv or any(arg in ("-h", "--help") for arg in argv):
27
- print(USAGE)
28
- sys.argv = [sys.argv[0], "--help"]
29
- with plugin_errors():
30
- cli(narrow_config(ServeConfig, argv))
31
- return
32
- legacy_id = any(a == "--legacy.id" or a.startswith("--legacy.id=") for a in argv)
33
- # An env-block flag (or a since-moved flat axis) skips the usage gate so the typed
34
- # parse renders its did-you-mean instead of a bare usage line.
35
- typed_axis = any(
36
- a.startswith(("--env.", "--taskset.", "--harness.", "--serve.")) for a in argv
37
- )
38
- if (
39
- not extract_id(argv, "env.taskset")
40
- and not legacy_id
41
- and not references_config_file(argv)
42
- and not typed_axis
43
- ):
44
- raise SystemExit(
45
- USAGE
46
- ) # need a taskset (positional / --env.taskset.id), a v0 --legacy.id, or @ file.toml
47
-
48
- with plugin_errors():
49
- config_type = narrow_config(ServeConfig, argv)
50
- sys.argv = [sys.argv[0], *argv]
51
- config = cli(config_type)
52
- if config.dry_run:
53
- print(config.model_dump_json(indent=2, exclude_none=True))
54
- return
55
- level = "DEBUG" if config.verbose else "INFO"
56
- setup_logging(level)
57
-
58
- # The pool config decides in-process vs router + worker pool (static or elastic); the
59
- # frontend speaks the same protocol either way. serve_env owns the SIGTERM teardown.
60
- # Pool workers are spawned with no logging, so hand serve_env the same setup to apply
61
- # in each one.
62
- server_kwargs = (
63
- {
64
- "env_id": config.legacy.id,
65
- "env_args": config.legacy.args,
66
- "extra_env_kwargs": config.legacy.extra_env_kwargs,
67
- }
68
- if config.is_legacy
69
- # `--serve.max-concurrent` is each v1 worker's episode bound; the legacy
70
- # bridge has never had one.
71
- else {"config": config.env, "max_concurrent": config.serve.max_concurrent}
72
- )
73
- serve_env(
74
- **pool_serve_kwargs(config.serve.pool),
75
- legacy=config.is_legacy,
76
- address=config.serve.address,
77
- log_setup=partial(setup_logging, level),
78
- **server_kwargs,
79
- )
@@ -1,62 +0,0 @@
1
- """Environment-server CLI configuration."""
2
-
3
- from pydantic import AliasChoices, Field, SerializeAsAny, model_validator
4
- from pydantic_config import BaseConfig
5
-
6
- from verifiers.v1.configs.cli.env import narrowed_env_annotation, resolve_env_field
7
- from verifiers.v1.configs.env import EnvConfig
8
- from verifiers.v1.configs.legacy import LegacyEnvConfig
9
- from verifiers.v1.configs.serve import ServingConfig
10
- from verifiers.v1.envs.single_agent import SingleAgentEnvConfig
11
-
12
-
13
- class ServeConfig(BaseConfig):
14
- """`uv run serve`: what to serve (`[env]`, or `[legacy]` for a classic v0 env) and
15
- how it's hosted (`[serve]`)."""
16
-
17
- env: SerializeAsAny[EnvConfig] = SingleAgentEnvConfig()
18
- """The environment — which env, its seed taskset, each agent, its knobs. Narrowed to
19
- the selected env's config class by the env id, else the taskset id."""
20
- serve: ServingConfig = ServingConfig()
21
- """How it's served: the worker pool, the bind address, each worker's episode bound."""
22
- legacy: LegacyEnvConfig = LegacyEnvConfig()
23
- """A classic (v0) environment to serve through the bridge instead of `[env]`."""
24
- verbose: bool = Field(False, validation_alias=AliasChoices("verbose", "v"))
25
- """Log at debug level instead of info."""
26
- dry_run: bool = False
27
- """Resolve + validate the config and dump it, then exit."""
28
-
29
- @property
30
- def is_legacy(self) -> bool:
31
- """Whether this run goes through the v0 bridge: a legacy id and no v1 taskset."""
32
- return self.legacy.id is not None and not self.env.taskset.id
33
-
34
- @property
35
- def env_id(self) -> str:
36
- """The run's identifier: the v1 env's, else the v0 env id."""
37
- return self.env.env_id or self.legacy.id or ""
38
-
39
- @model_validator(mode="after")
40
- def _refuse_mixed_run(self):
41
- # A v0 id next to any v1 env identity leaves one of the two going nowhere, and
42
- # which one depends on `is_legacy`: a taskset makes it False, so the v0 env never
43
- # loads; a bare `--env.id` leaves it True, so the v0 env runs under the v1 name.
44
- if self.legacy.id is None or not self.env.env_id:
45
- return self
46
- if self.env.taskset.id:
47
- raise ValueError(
48
- f"--legacy.id {self.legacy.id!r} is a classic (v0) env and can't combine "
49
- f"with the v1 taskset {self.env.taskset.id!r}. Pairing a reusable env with "
50
- f"a taskset is --env.id {self.legacy.id!r} (TOML: id under [env]); to run "
51
- "the v0 env instead, drop the taskset."
52
- )
53
- raise ValueError(
54
- f"--legacy.id {self.legacy.id!r} is a classic (v0) env and can't combine with "
55
- f"the v1 env --env.id {self.env.id!r}: the v0 env is what would run, stamped "
56
- "with the v1 env's name. Keep whichever one you meant to run."
57
- )
58
-
59
- @model_validator(mode="before")
60
- @classmethod
61
- def _resolve_env(cls, data):
62
- return resolve_env_field(data, narrowed_env_annotation(cls))