verifiers 0.2.2.dev73__py3-none-any.whl → 0.2.2.dev75__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
verifiers/v1/agent.py CHANGED
@@ -29,7 +29,7 @@ from verifiers.v1.runtimes import (
29
29
  Runtime,
30
30
  RuntimeConfig,
31
31
  SubprocessConfig,
32
- make_runtime,
32
+ provision_runtime,
33
33
  runtime_is_local,
34
34
  )
35
35
  from verifiers.v1.session import RolloutLimits
@@ -547,14 +547,8 @@ class Agent:
547
547
  if task is not None
548
548
  else self.runtime_config
549
549
  )
550
- runtime = make_runtime(config)
551
- try:
552
- # start() inside the try: a failed start may already hold a remote
553
- # sandbox, so it must reach stop() (safe on a partially-started runtime).
554
- await runtime.start()
550
+ async with provision_runtime(config) as runtime:
555
551
  yield runtime
556
- finally:
557
- await runtime.stop()
558
552
 
559
553
 
560
554
  class _EpisodeAgent(Agent):
@@ -332,8 +332,8 @@ def Progress(
332
332
 
333
333
  def _score_segments(traces: list[Trace], source: str) -> str | None:
334
334
  """`name mean` segments for every reward/metric key seen across `traces`,
335
- first-seen order (a trace records only the functions that ran for it, so keys
336
- can vary); None when no trace recorded anything."""
335
+ first-seen order (a trace carries only its own task's signals — unscored ones
336
+ seeded as `None` — so keys can vary); None when no trace recorded anything."""
337
337
  names: list[str] = []
338
338
  for trace in traces:
339
339
  names.extend(n for n in getattr(trace, source) if n not in names)
@@ -347,11 +347,13 @@ def _score_segments(traces: list[Trace], source: str) -> str | None:
347
347
 
348
348
 
349
349
  def _score(trace: Trace, source: str, name: str) -> float:
350
- """Rewards carry raw score + weight; the breakdown shows the raw score."""
350
+ """Rewards carry raw score + weight; the breakdown shows the raw score. A missing
351
+ or unscored (seeded `None`) entry reads as 0.0."""
351
352
  if source == "rewards":
352
353
  reward = trace.rewards.get(name)
353
354
  return reward.score if reward is not None else 0.0
354
- return trace.metrics.get(name, 0.0)
355
+ value = trace.metrics.get(name)
356
+ return value if value is not None else 0.0
355
357
 
356
358
 
357
359
  def _breakdown(scored: list[Trace], done: list[Trace]) -> Table | None:
@@ -398,7 +398,8 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
398
398
  solution.record_metric(f"judge/{criterion.name}", scores[criterion.name])
399
399
  if self.config.score.task_weight != 1.0:
400
400
  for reward in solution.rewards.values():
401
- reward.weight *= self.config.score.task_weight
401
+ if reward is not None:
402
+ reward.weight *= self.config.score.task_weight
402
403
  total = sum(criterion.weight for criterion in criteria)
403
404
  reward = sum(c.weight * scores[c.name] for c in criteria) / total
404
405
  solution.record_reward("judge", reward, weight=self.config.score.judge_weight)
verifiers/v1/harness.py CHANGED
@@ -12,7 +12,12 @@ from verifiers.v1.errors import HarnessError, SandboxError, boundary
12
12
  from verifiers.v1.runtimes import ProgramResult, Runtime
13
13
  from verifiers.v1.task import TaskData
14
14
  from verifiers.v1.types import Messages
15
- from verifiers.v1.utils.decorators import discover_decorated, invoke_all
15
+ from verifiers.v1.utils.decorators import (
16
+ discover_decorated,
17
+ invoke_all,
18
+ seed,
19
+ unseed,
20
+ )
16
21
 
17
22
  if TYPE_CHECKING:
18
23
  # Annotation-only: `Trace` appears in signatures only, so this module stays
@@ -155,10 +160,12 @@ class Harness(ABC, Generic[ConfigT]):
155
160
  `runtime`) and can read what the harness left behind in the runtime."""
156
161
  available = {"task": trace.task.data, "trace": trace, "runtime": runtime}
157
162
  fns = discover_decorated(self, "metric")
163
+ seed(trace.metrics, (fn.__name__ for fn in fns))
158
164
  async with boundary(HarnessError, f"harness {self.config.id!r} metric"):
159
165
  results = await invoke_all(fns, available)
160
166
  for fn, result in zip(fns, results):
161
167
  if isinstance(result, Mapping):
168
+ unseed(trace.metrics, fn.__name__)
162
169
  trace.record_metrics(result)
163
170
  else:
164
171
  trace.record_metric(fn.__name__, result)
@@ -1,3 +1,5 @@
1
+ from collections.abc import AsyncIterator
2
+ from contextlib import asynccontextmanager
1
3
  from typing import Annotated
2
4
 
3
5
  from pydantic import Field
@@ -45,6 +47,22 @@ def make_runtime(config: RuntimeConfig, name: str | None = None) -> Runtime:
45
47
  return runtime
46
48
 
47
49
 
50
+ @asynccontextmanager
51
+ async def provision_runtime(
52
+ config: RuntimeConfig, name: str | None = None
53
+ ) -> AsyncIterator[Runtime]:
54
+ """Provision a box from `config` and tear it down on exit.
55
+
56
+ `start()` sits inside the `try`: a failed start may already hold a paid sandbox, so
57
+ it has to reach `stop()` (which is safe on a partially-started runtime)."""
58
+ runtime = make_runtime(config, name)
59
+ try:
60
+ await runtime.start()
61
+ yield runtime
62
+ finally:
63
+ await runtime.stop()
64
+
65
+
48
66
  def runtime_is_local(config: RuntimeConfig) -> bool:
49
67
  """Whether a runtime of this config exchanges host-local URLs without a public
50
68
  tunnel, read off the runtime class without provisioning one."""
@@ -71,5 +89,6 @@ __all__ = [
71
89
  "SubprocessRuntime",
72
90
  "SubprocessRuntimeInfo",
73
91
  "make_runtime",
92
+ "provision_runtime",
74
93
  "runtime_is_local",
75
94
  ]
@@ -2,6 +2,7 @@
2
2
 
3
3
  import asyncio
4
4
  import atexit
5
+ import base64
5
6
  import contextlib
6
7
  import hashlib
7
8
  import logging
@@ -285,9 +286,42 @@ class Runtime(ABC):
285
286
  argv = await self.prepare_uv_script(script, env)
286
287
  return await self.run([*argv, *(args or [])], env or {})
287
288
 
289
+ async def read(self, path: str, max_bytes: int | None = None) -> bytes:
290
+ """Read `path` into host memory. `max_bytes` caps the transfer, raising past
291
+ the cap — for a file written by something we don't control, whose size we
292
+ can't assume. The cap is enforced inside the box rather than after the
293
+ transfer, and base64 because `run` returns decoded text. Framework method —
294
+ override `_read`, not this."""
295
+ if max_bytes is None:
296
+ return await self._read(path)
297
+ # Through a temp file, not a pipe: `head | base64` exits with base64's 0
298
+ # even when the path is missing, and a missing file must raise here just
299
+ # as it does from `_read`.
300
+ result = await self.run(
301
+ [
302
+ "sh",
303
+ "-c",
304
+ (
305
+ "t=$(mktemp) || exit 1; "
306
+ 'head -c "$1" -- "$2" > "$t" || { rm -f "$t"; exit 1; }; '
307
+ 'base64 < "$t"; rc=$?; rm -f "$t"; exit $rc'
308
+ ),
309
+ "sh",
310
+ str(max_bytes + 1),
311
+ path,
312
+ ],
313
+ {},
314
+ )
315
+ if result.exit_code:
316
+ raise SandboxError(f"read {path!r}: {result.stderr.strip()[-500:]}")
317
+ data = base64.b64decode(result.stdout)
318
+ if len(data) > max_bytes:
319
+ raise SandboxError(f"read {path!r}: over the {max_bytes} byte limit")
320
+ return data
321
+
288
322
  @abstractmethod
289
- async def read(self, path: str) -> bytes:
290
- pass
323
+ async def _read(self, path: str) -> bytes:
324
+ """Read the whole file at `path`; `read` adds the optional transfer cap."""
291
325
 
292
326
  @abstractmethod
293
327
  async def write(self, path: str, data: bytes) -> None:
@@ -334,7 +334,7 @@ class DockerRuntime(Runtime):
334
334
  if run.exit_code != 0:
335
335
  raise SandboxError(f"docker exec -d failed: {run.stderr.strip()}")
336
336
 
337
- async def read(self, path: str) -> bytes:
337
+ async def _read(self, path: str) -> bytes:
338
338
  proc = await asyncio.create_subprocess_exec(
339
339
  "docker",
340
340
  "exec",
@@ -165,7 +165,7 @@ class ModalRuntime(Runtime):
165
165
  return path
166
166
  return f"{self.config.workdir.rstrip('/')}/{path}"
167
167
 
168
- async def read(self, path: str) -> bytes:
168
+ async def _read(self, path: str) -> bytes:
169
169
  try:
170
170
  return await self._sandbox.filesystem.read_bytes.aio(self._abs(path))
171
171
  except Exception as e:
@@ -270,7 +270,7 @@ class PrimeRuntime(Runtime):
270
270
  f"prime background launch failed: {result.stderr.strip()}"
271
271
  )
272
272
 
273
- async def read(self, path: str) -> bytes:
273
+ async def _read(self, path: str) -> bytes:
274
274
  # Avoid background-job log limits and base64 overhead by downloading binary data directly.
275
275
  # The temporary file is removed on every exit, and its byte read stays off the event loop.
276
276
  target = (
@@ -96,7 +96,7 @@ class SubprocessRuntime(Runtime):
96
96
  proc
97
97
  ) # killed in stop() — a host process won't die on its own
98
98
 
99
- async def read(self, path: str) -> bytes:
99
+ async def _read(self, path: str) -> bytes:
100
100
  return await asyncio.to_thread((self.workdir / path).read_bytes)
101
101
 
102
102
  async def write(self, path: str, data: bytes) -> None:
verifiers/v1/task.py CHANGED
@@ -23,7 +23,12 @@ from verifiers.v1.errors import TaskError, boundary
23
23
  from verifiers.v1.state import StateT
24
24
  from verifiers.v1.types import Messages, content_text
25
25
  from verifiers.v1.utils.artifacts import Artifact
26
- from verifiers.v1.utils.decorators import discover_decorated, invoke_all
26
+ from verifiers.v1.utils.decorators import (
27
+ discover_decorated,
28
+ invoke_all,
29
+ seed,
30
+ unseed,
31
+ )
27
32
  from verifiers.v1.utils.generic import concrete_type
28
33
 
29
34
  if TYPE_CHECKING:
@@ -180,31 +185,36 @@ class Task(Generic[DataT, StateT, ConfigT]):
180
185
  judge for judge in judges if not requires_runtime(judge.score)
181
186
  ]
182
187
 
188
+ seed(trace.metrics, (fn.__name__ for fn in metrics))
189
+ seed(trace.rewards, (fn.__name__ for fn in rewards))
190
+ seed(trace.rewards, (judge.reward_name for judge in judges))
191
+
183
192
  metric_results = await invoke_all(metrics, available)
184
193
  for fn, result in zip(metrics, metric_results):
185
194
  if isinstance(result, Mapping):
195
+ unseed(trace.metrics, fn.__name__)
186
196
  trace.record_metrics(result)
187
197
  else:
188
198
  trace.record_metric(fn.__name__, result)
189
199
  reward_results = await invoke_all(rewards, available)
190
200
  for fn, result in zip(rewards, reward_results):
191
201
  weight = getattr(fn, "_vf_weight", 1.0)
192
- items = (
193
- result.items()
194
- if isinstance(result, Mapping)
195
- else [(fn.__name__, result)]
196
- )
202
+ if isinstance(result, Mapping):
203
+ unseed(trace.rewards, fn.__name__)
204
+ items = result.items()
205
+ else:
206
+ items = [(fn.__name__, result)]
197
207
  for key, value in items:
198
208
  trace.record_reward(key, value, weight)
199
209
  judge_results = await invoke_all(
200
210
  [judge.score for judge in judges], available
201
211
  )
202
212
  for judge, result in zip(judges, judge_results):
203
- items = (
204
- result.items()
205
- if isinstance(result, Mapping)
206
- else [(judge.reward_name, result)]
207
- )
213
+ if isinstance(result, Mapping):
214
+ unseed(trace.rewards, judge.reward_name)
215
+ items = result.items()
216
+ else:
217
+ items = [(judge.reward_name, result)]
208
218
  for key, value in items:
209
219
  trace.record_reward(key, value, judge.config.weight)
210
220
 
@@ -1,3 +1,4 @@
1
+ from verifiers.v1.tasksets.harbor.env import HarborEnv, HarborEnvConfig
1
2
  from verifiers.v1.tasksets.harbor.taskset import (
2
3
  HarborConfig,
3
4
  HarborData,
@@ -5,4 +6,11 @@ from verifiers.v1.tasksets.harbor.taskset import (
5
6
  HarborTaskset,
6
7
  )
7
8
 
8
- __all__ = ["HarborConfig", "HarborData", "HarborTask", "HarborTaskset"]
9
+ __all__ = [
10
+ "HarborConfig",
11
+ "HarborData",
12
+ "HarborEnv",
13
+ "HarborEnvConfig",
14
+ "HarborTask",
15
+ "HarborTaskset",
16
+ ]
@@ -0,0 +1,124 @@
1
+ """The harbor taskset's own env: the single solver seat, plus separate-verifier
2
+ grading for tasks that declare ``[verifier].environment_mode = "separate"``.
3
+
4
+ The default env for harbor runs (the taskset package exports it). A shared-verifier
5
+ task runs exactly as under the single-agent env: one `agent` trace, graded in the
6
+ box it worked in. A separate-verifier task is graded by `finalize` instead: the
7
+ solver's declared artifacts travel (collected by its task `finalize` while its box
8
+ is alive), a fresh box is provisioned from the task's verifier declaration,
9
+ `tests/` is staged there, and the verifier's rewards land on the solver's trace.
10
+ No second agent is involved — the verifier is the task's own `tests/test.sh`.
11
+ """
12
+
13
+ import asyncio
14
+ import logging
15
+ from contextlib import AsyncExitStack
16
+
17
+ from pydantic import Field
18
+
19
+ import verifiers.v1 as vf
20
+ from verifiers.v1.runtimes import RuntimeConfig, provision_runtime
21
+ from verifiers.v1.tasksets.harbor.taskset import (
22
+ HarborTask,
23
+ verifier_box_data,
24
+ )
25
+ from verifiers.v1.utils.artifacts import restore
26
+ from verifiers.v1.utils.compile import resolve_runtime_config
27
+ from verifiers.v1.utils.retries import backoff
28
+
29
+ logger = logging.getLogger(__name__)
30
+
31
+
32
+ class HarborEnvConfig(vf.EnvConfig):
33
+ agent: vf.AgentConfig = vf.AgentConfig()
34
+ """The one seat — the policy under evaluation/training; pin
35
+ `--env.agent.harness.*` to choose its program or runtime."""
36
+ verifier_runtime: RuntimeConfig | None = None
37
+ """Where a separate-verifier task grades. None derives the grading box from
38
+ the solver's runtime policy; set it (e.g. `--env.verifier-runtime.type prime
39
+ --env.verifier-runtime.vm true`) when the verifier needs different placement
40
+ than the agent."""
41
+ verifier_retries: int = Field(2, ge=0)
42
+ """Extra attempts at provisioning-and-grading the separate box before the
43
+ episode fails. Grading is deterministic; what these retry is the
44
+ infrastructure around it (image pulls, provisioning)."""
45
+
46
+
47
+ class HarborEnv(vf.Env[HarborEnvConfig]):
48
+ async def run(self, task: vf.Task, agents: vf.Agents) -> None:
49
+ if not isinstance(task, HarborTask):
50
+ raise TypeError(
51
+ f"the harbor env runs harbor tasks; got {type(task).__name__}"
52
+ )
53
+ if task.data.verifier is None:
54
+ await agents.agent.run(task)
55
+ return
56
+ # Resolve the verifier's box before the solve, so an impossible pairing
57
+ # (e.g. a restricted Prime verifier without vm=true) costs nothing
58
+ # rather than a full agent run.
59
+ self._verifier_config(task)
60
+ await agents.agent.run(task.graded_elsewhere())
61
+
62
+ def _verifier_config(self, task: HarborTask) -> RuntimeConfig:
63
+ base = (
64
+ self.config.verifier_runtime
65
+ if self.config.verifier_runtime is not None
66
+ else self.config.agent.runtime
67
+ )
68
+ return resolve_runtime_config(base, HarborTask(verifier_box_data(task.data)))
69
+
70
+ async def finalize(self, task: vf.Task, episode: vf.Episode) -> None:
71
+ """Grade a separate-verifier task in its own box, onto the solver's trace.
72
+
73
+ Provision a fresh box from the task's verifier declaration, restore the
74
+ solver's collected artifacts, stage `tests/`, run the verifier, and record
75
+ its rewards (and any extra reward.json keys as metrics) on the solver's
76
+ trace. Infrastructure failures retry per `verifier_retries`; the last one
77
+ fails the episode — a grading box that can't be reached must never read
78
+ as reward 0."""
79
+ if not isinstance(task, HarborTask) or task.data.verifier is None:
80
+ return
81
+ solution = episode.traces[0]
82
+ if not solution.ok:
83
+ return
84
+ grader = HarborTask(verifier_box_data(task.data))
85
+ scores = await self._grade(self._verifier_config(task), grader, solution)
86
+ items = scores.items() if isinstance(scores, dict) else [("solved", scores)]
87
+ for name, value in items:
88
+ solution.record_reward(name, value)
89
+
90
+ async def _grade(
91
+ self, config: RuntimeConfig, grader: HarborTask, solution: vf.Trace
92
+ ) -> float | dict[str, float]:
93
+ last: Exception | None = None
94
+ for attempt in range(self.config.verifier_retries + 1):
95
+ if attempt:
96
+ delay = backoff(attempt - 1)
97
+ logger.warning(
98
+ "harbor verifier attempt %d/%d failed (%s); retrying in %.1fs",
99
+ attempt,
100
+ self.config.verifier_retries + 1,
101
+ last,
102
+ delay,
103
+ )
104
+ await asyncio.sleep(delay)
105
+ try:
106
+ # The scoring deadline covers provisioning and grading, but not the
107
+ # box's teardown: a score already in hand must not be discarded
108
+ # because the teardown ran out the clock.
109
+ async with AsyncExitStack() as boxes:
110
+ async with asyncio.timeout(grader.data.timeout.scoring):
111
+ box = await boxes.enter_async_context(provision_runtime(config))
112
+ await box.prepare_setup()
113
+ # Artifacts first, tests second: an artifact entry pointing
114
+ # into /tests must not survive staging, which wipes and
115
+ # rebuilds that directory.
116
+ await restore(box, solution.state.artifacts)
117
+ await grader._stage_tests(box, wipe=True)
118
+ await box.prepare_execution([])
119
+ scores = await grader._graded(box, solution)
120
+ return scores
121
+ except Exception as e: # noqa: BLE001 - each attempt's failure is retried
122
+ last = e
123
+ assert last is not None
124
+ raise last
@@ -1,18 +1,25 @@
1
1
  """Harbor tasksets backed by Harbor Hub packages.
2
2
 
3
3
  The Harbor CLI downloads and caches each task directory. Its verifier runs in the
4
- same runtime the harness edited, then writes the score to
4
+ runtime the harness edited — or, when the task asks for it with
5
+ ``[verifier].environment_mode = "separate"``, in a second box the agent never
6
+ touched, carrying only what the task declared — the harbor env provisions and
7
+ grades that box (see ``env.py``). Either way the score lands in
5
8
  ``/logs/verifier/reward.json`` or the legacy ``reward.txt``.
6
9
 
7
10
  A pullable ``[environment].docker_image`` becomes ``TaskData.image``. Verifiers does
8
11
  not build Dockerfile-only environments, so those are rejected unless ``ignore_dockerfile``
9
12
  deliberately uses the harness runtime image. Tasks without an environment also use that
10
- image unless ``require_image`` is set.
13
+ image unless ``require_image`` is set. The same rule applies to a declared
14
+ ``[verifier.environment]``: it needs a pullable ``docker_image``, since Harbor would
15
+ otherwise build the verifier image from ``tests/Dockerfile``.
11
16
  """
12
17
 
13
18
  import asyncio
19
+ import copy
14
20
  import hashlib
15
21
  import io
22
+ import logging
16
23
  import shutil
17
24
  import subprocess
18
25
  import sys
@@ -26,7 +33,7 @@ from typing import Annotated
26
33
  from pydantic import BaseModel, ConfigDict, Field, TypeAdapter, ValidationError
27
34
 
28
35
  from verifiers.v1.configs.taskset import TasksetConfig
29
- from verifiers.v1.errors import SandboxError
36
+ from verifiers.v1.errors import SandboxError, TaskError
30
37
  from verifiers.v1.runtimes import Runtime
31
38
  from verifiers.v1.task import Task, TaskData, TaskResources, TaskTimeout
32
39
  from verifiers.v1.taskset import Taskset
@@ -34,9 +41,12 @@ from verifiers.v1.trace import Trace
34
41
  from verifiers.v1.utils.artifacts import Artifact, collect
35
42
  from verifiers.v1.utils.decorators import reward
36
43
 
44
+ logger = logging.getLogger(__name__)
45
+
37
46
  CACHE = Path.home() / ".cache" / "harbor"
38
47
  HARBOR_INSTALL_HINT = "uv sync --python 3.12 --extra harbor"
39
48
  REWARD_JSON = "/logs/verifier/reward.json"
49
+ MAX_REWARD_BYTES = 1024 * 1024
40
50
  REWARD_JSON_ADAPTER = TypeAdapter(
41
51
  float | Annotated[dict[str, float], Field(min_length=1)],
42
52
  config=ConfigDict(strict=True, allow_inf_nan=False),
@@ -76,6 +86,11 @@ class HarborConfig(TasksetConfig):
76
86
  instead of rejecting it. The Dockerfile is NOT built, so the task scores against the
77
87
  harness image rather than its declared environment — only correct when that image already
78
88
  has what the task needs (e.g. you've pointed the runtime at the right image)."""
89
+ ignore_separate_verifier: bool = False
90
+ """Grade every task in the agent's own box, even one whose `[verifier]` asks for a
91
+ separate one. Trades the isolation for a sandbox per task; useful when provisioning
92
+ is the bottleneck. Note what it gives up: the grader becomes reachable by the agent
93
+ that just ran."""
79
94
 
80
95
 
81
96
  class Author(BaseModel):
@@ -90,6 +105,27 @@ class CollectHook(BaseModel):
90
105
  timeout_sec: float = 600.0
91
106
 
92
107
 
108
+ class VerifierConfig(BaseModel):
109
+ """The box this task's verifier wants, when it wants one of its own.
110
+
111
+ `None` on `HarborData` means shared — grade where the agent worked, which is still
112
+ Harbor's default and every task that says nothing."""
113
+
114
+ image: str | None = None
115
+ """Pullable ref from `[verifier.environment].docker_image`. None keeps the task's
116
+ own image, which is what Harbor's fresh copy of `[environment]` resolves to."""
117
+ resources: TaskResources = TaskResources()
118
+ workdir: str | None = None
119
+ fresh_copy: bool = False
120
+ """Whether this came from Harbor's fresh copy of `[environment]` rather than a
121
+ declared `[verifier.environment]`. A fresh copy inherits the agent box's resolved
122
+ resources; a declared environment states its own, and what it omits falls back to
123
+ the run's rather than to the agent's task-derived values."""
124
+ network_allow: list[str] = Field(default_factory=lambda: ["*"])
125
+ """Destinations the verifier may reach, from the verifier's network mode. `["*"]`
126
+ is unrestricted; `[]` is Harbor's `no-network` / `allow_internet = false`."""
127
+
128
+
93
129
  class HarborData(TaskData):
94
130
  """Parsed ``task.toml`` metadata plus the host-side verifier directory.
95
131
 
@@ -111,6 +147,9 @@ class HarborData(TaskData):
111
147
  collect: list[CollectHook] = Field(default_factory=list)
112
148
  """`[[verifier.collect]]` blocks: commands that snapshot runtime state into files
113
149
  after the agent stops, so the files can travel to a grading box as artifacts."""
150
+ verifier: VerifierConfig | None = None
151
+ """The verifier's own box, when `[verifier].environment_mode` asks for one. None
152
+ grades in the agent's box."""
114
153
 
115
154
 
116
155
  class HarborTask(Task[HarborData]):
@@ -145,30 +184,59 @@ class HarborTask(Task[HarborData]):
145
184
  )
146
185
  trace.state.artifacts = await collect(runtime, self.data.artifacts)
147
186
 
148
- @reward(weight=1.0)
149
- async def solved(self, runtime: Runtime, trace: Trace) -> float | dict[str, float]:
187
+ def graded_elsewhere(self) -> "HarborTask":
188
+ """A copy whose `solved` records nothing here: the harbor env grades this
189
+ task's finished work in a separate box of the task's choosing."""
190
+ clone = copy.copy(self)
191
+ clone._graded_elsewhere = True
192
+ return clone
193
+
194
+ _graded_elsewhere: bool = False
195
+
196
+ async def _stage_tests(self, runtime: Runtime, wipe: bool = False) -> None:
197
+ """Put the task package's `tests/` in `/tests`, where `test.sh` expects it.
198
+
199
+ Raises rather than scoring stale state: a leftover reward file — planted by
200
+ the agent or shipped in the image — must be gone before `test.sh` runs, so a
201
+ removal that fails must not fall through to reading it.
202
+
203
+ `wipe` for a box we did not watch being built: a fresh container of the task's
204
+ image can ship its own `/tests`, and a leftover file there would be graded as
205
+ though it came from the package.
206
+ """
150
207
  await runtime.write(
151
208
  "/tmp/tests.tgz", make_tar(Path(self.data.task_dir) / "tests")
152
209
  )
153
- await runtime.run(
154
- [
155
- "sh",
156
- "-c",
157
- "mkdir -p /logs/verifier /tests && tar -xzf /tmp/tests.tgz -C /tests",
158
- ],
159
- {},
160
- )
161
- await runtime.run(
162
- [
163
- "sh",
164
- "-c",
165
- (
166
- "rm -f /logs/verifier/reward.json /logs/verifier/reward.txt"
167
- " && cd /tests && bash test.sh"
168
- ),
169
- ],
170
- verifier_env(self.data),
210
+ stage = (
211
+ f"{'rm -rf /tests && ' if wipe else ''}"
212
+ "rm -f /logs/verifier/reward.json /logs/verifier/reward.txt && "
213
+ "mkdir -p /logs/verifier /tests && tar -xzf /tmp/tests.tgz -C /tests"
171
214
  )
215
+ result = await runtime.run(["sh", "-c", stage], {})
216
+ if result.exit_code:
217
+ raise TaskError(
218
+ f"staging tests failed (exit {result.exit_code}): "
219
+ f"{(result.stderr or result.stdout).strip()[-500:]}"
220
+ )
221
+
222
+ @reward(weight=1.0)
223
+ async def solved(self, runtime: Runtime, trace: Trace) -> float | dict[str, float]:
224
+ if self.data.verifier is not None:
225
+ if not self._graded_elsewhere:
226
+ raise TaskError(
227
+ f"task {self.data.name!r} declares a separate verifier "
228
+ '([verifier].environment_mode = "separate"); grade it through '
229
+ "the harbor env (this taskset's default), or force shared "
230
+ "grading with --taskset.ignore-separate-verifier"
231
+ )
232
+ return {}
233
+ await self._stage_tests(runtime)
234
+ return await self._graded(runtime, trace)
235
+
236
+ async def _graded(self, runtime: Runtime, trace: Trace) -> float | dict[str, float]:
237
+ # By absolute path, in the runtime's configured workdir: Harbor execs the
238
+ # script the same way, and scripts do grade the agent's work at `$PWD`.
239
+ await runtime.run(["bash", "/tests/test.sh"], verifier_env(self.data))
172
240
  scores = await self._reward_json(runtime)
173
241
  if scores is not None:
174
242
  if isinstance(scores, dict) and "reward" in scores:
@@ -178,19 +246,75 @@ class HarborTask(Task[HarborData]):
178
246
  return {"reward": scores["reward"]}
179
247
  return scores
180
248
  try:
181
- reward = (await runtime.read("/logs/verifier/reward.txt")).decode().strip()
249
+ reward = (
250
+ (
251
+ await runtime.read(
252
+ "/logs/verifier/reward.txt", max_bytes=MAX_REWARD_BYTES
253
+ )
254
+ )
255
+ .decode()
256
+ .strip()
257
+ )
182
258
  return float(reward or 0)
183
259
  except (SandboxError, OSError, ValueError):
184
260
  return 0.0
185
261
 
186
262
  async def _reward_json(self, runtime: Runtime) -> float | dict[str, float] | None:
187
- """Read Harbor's scalar or keyed JSON reward, if it is valid."""
263
+ """Read Harbor's scalar or keyed JSON reward, if it is valid.
264
+
265
+ Bounded: this is a grading input, and nothing guarantees its size.
266
+ """
188
267
  try:
189
- return REWARD_JSON_ADAPTER.validate_json(await runtime.read(REWARD_JSON))
268
+ return REWARD_JSON_ADAPTER.validate_json(
269
+ await runtime.read(REWARD_JSON, max_bytes=MAX_REWARD_BYTES)
270
+ )
190
271
  except (SandboxError, OSError, ValidationError):
191
272
  return None
192
273
 
193
274
 
275
+ def verifier_box_data(data: HarborData) -> HarborData:
276
+ """The verifier's box, declared as task data — the harbor env resolves the
277
+ grading runtime from it (image, workdir, resources, network policy), exactly
278
+ as the solver's box resolves from the solver task's.
279
+
280
+ Which box follows Harbor: a declared `[verifier.environment]` states its own
281
+ image, workdir, and resources, and what it omits is the run's default; a
282
+ fresh copy of `[environment]` keeps the task's own. The verifier's network
283
+ policy applies either way."""
284
+ verifier = data.verifier
285
+ if verifier is None:
286
+ raise TaskError(f"task {data.name!r} declares no separate verifier")
287
+ fresh = verifier.fresh_copy
288
+ return data.model_copy(
289
+ update={
290
+ "name": f"{data.name} (verifier)",
291
+ "image": verifier.image if verifier.image is not None else data.image,
292
+ "workdir": data.workdir if fresh else verifier.workdir,
293
+ "resources": data.resources if fresh else verifier.resources,
294
+ "network_allow": list(verifier.network_allow),
295
+ "network_block": [],
296
+ }
297
+ )
298
+
299
+
300
+ def task_resources(environment, multiplier: float) -> TaskResources:
301
+ """Harbor environment resource requests, scaled, as `TaskResources`.
302
+
303
+ Harbor declares CPU counts and MB sizes; `TaskResources` wants counts and GB.
304
+ GPU requests are never scaled.
305
+ """
306
+ return TaskResources(
307
+ cpu=environment.cpus * multiplier if environment.cpus else None,
308
+ memory=environment.memory_mb / 1024 * multiplier
309
+ if environment.memory_mb
310
+ else None,
311
+ gpu=str(environment.gpus) if environment.gpus else None,
312
+ disk=environment.storage_mb / 1024 * multiplier
313
+ if environment.storage_mb
314
+ else None,
315
+ )
316
+
317
+
194
318
  def harbor_cli() -> str:
195
319
  scripts_dir = Path(sys.executable).parent
196
320
  harbor_bin = shutil.which("harbor", path=str(scripts_dir))
@@ -318,7 +442,7 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
318
442
 
319
443
  harbor_task = HarborModelTask(task_dir)
320
444
  parsed = harbor_task.config
321
- artifacts, collect = parse_verifier_extras(task_dir, parsed)
445
+ artifacts, hooks, verifier = parse_verifier_extras(task_dir, parsed, harbor_config)
322
446
  environment = parsed.environment
323
447
  network = parsed.agent.explicit_phase_policy() or environment.resolve_baseline()
324
448
  task, meta = parsed.task, parsed.metadata
@@ -368,18 +492,7 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
368
492
  if scoring_timeout is not None
369
493
  else None,
370
494
  ),
371
- resources=TaskResources(
372
- cpu=environment.cpus * harbor_config.resource_multiplier
373
- if environment.cpus
374
- else None,
375
- memory=environment.memory_mb / 1024 * harbor_config.resource_multiplier
376
- if environment.memory_mb
377
- else None,
378
- gpu=str(environment.gpus) if environment.gpus else None,
379
- disk=environment.storage_mb / 1024 * harbor_config.resource_multiplier
380
- if environment.storage_mb
381
- else None,
382
- ),
495
+ resources=task_resources(environment, harbor_config.resource_multiplier),
383
496
  keywords=task.keywords if task else [],
384
497
  authors=authors,
385
498
  difficulty=meta.get("difficulty"),
@@ -388,14 +501,22 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
388
501
  task_dir=str(task_dir),
389
502
  verifier_env=parsed.verifier.env,
390
503
  artifacts=artifacts,
391
- collect=collect,
504
+ collect=hooks,
505
+ verifier=verifier,
392
506
  )
393
507
 
394
508
 
395
509
  def parse_verifier_extras(
396
- task_dir: Path, parsed
397
- ) -> tuple[list[Artifact], list[CollectHook]]:
398
- """Parse supported artifact and collect-hook settings."""
510
+ task_dir: Path, parsed, harbor_config: HarborConfig
511
+ ) -> tuple[list[Artifact], list[CollectHook], VerifierConfig | None]:
512
+ """Harbor's `artifacts`, `[[verifier.collect]]` blocks, and verifier environment,
513
+ narrowed to what verifiers' verifier-runtime integration can honor.
514
+
515
+ The convention dir is deliberately not prepended here (Harbor's
516
+ `with_convention_entry` would): collection injects it itself, as an optional sweep.
517
+ Prepending it would make it an explicitly declared entry, and declared entries are
518
+ required — which would fail every task that never writes there.
519
+ """
399
520
  from harbor.constants import MAIN_SERVICE_NAME
400
521
  from harbor.models.task.artifacts import (
401
522
  effective_artifact_service,
@@ -403,13 +524,6 @@ def parse_verifier_extras(
403
524
  )
404
525
 
405
526
  verifier = parsed.verifier
406
- if verifier.environment is not None:
407
- raise ValueError(
408
- f"{task_dir.name}: [verifier.environment] declares a separate verifier "
409
- "image. Grading runs in a fresh box built from the task's own image, so "
410
- "only the agent's delta has to travel; a different verifier image needs "
411
- "the full working tree copied over and isn't supported yet."
412
- )
413
527
  if verifier.user is not None:
414
528
  raise ValueError(f"{task_dir.name}: [verifier].user is not supported")
415
529
 
@@ -447,7 +561,89 @@ def parse_verifier_extras(
447
561
  )
448
562
  hooks.append(CollectHook(command=hook.command, timeout_sec=hook.timeout_sec))
449
563
 
450
- return artifacts, hooks
564
+ return artifacts, hooks, parse_verifier_environment(task_dir, parsed, harbor_config)
565
+
566
+
567
+ def parse_verifier_environment(
568
+ task_dir: Path, parsed, harbor_config: HarborConfig
569
+ ) -> VerifierConfig | None:
570
+ """The box Harbor wants this task's verifier in, or None to grade in the agent's.
571
+
572
+ Harbor resolves `[verifier.environment]` if declared, else a deep copy of
573
+ `[environment]` — so a mode-only `separate` lands on the task's own image and needs
574
+ nothing but a second box. A declared environment is the case that can name a
575
+ different image, and the case that can name none at all: there Harbor builds
576
+ `tests/Dockerfile`, which verifiers never does.
577
+ """
578
+ from harbor.models.task.config import NetworkMode, TaskOS
579
+ from harbor.models.task.verifier_mode import (
580
+ VerifierEnvironmentMode,
581
+ resolve_effective_verifier_env_config,
582
+ resolve_task_verifier_mode,
583
+ )
584
+
585
+ if resolve_task_verifier_mode(parsed) != VerifierEnvironmentMode.SEPARATE:
586
+ return None
587
+ if harbor_config.ignore_separate_verifier:
588
+ logger.warning(
589
+ "%s: asks for a separate verifier; grading in the agent's box anyway "
590
+ "(--taskset.ignore-separate-verifier)",
591
+ task_dir.name,
592
+ )
593
+ return None
594
+
595
+ environment = resolve_effective_verifier_env_config(parsed, None)
596
+ if environment is None: # unreachable while the mode is SEPARATE
597
+ raise ValueError(f"{task_dir.name}: separate verifier resolved no environment")
598
+ declared = parsed.verifier.environment is not None
599
+
600
+ if declared and environment.docker_image is None:
601
+ if not harbor_config.ignore_dockerfile:
602
+ raise ValueError(
603
+ f"{task_dir.name}: [verifier.environment] names no docker_image, so "
604
+ "Harbor would build the verifier image from tests/Dockerfile. Verifiers "
605
+ "pulls images and never builds them: build and push it yourself (e.g. "
606
+ "`prime images push`) and set [verifier.environment].docker_image to the "
607
+ "resulting ref, or pass --taskset.ignore-dockerfile to grade in the "
608
+ "agent's image instead."
609
+ )
610
+ logger.warning(
611
+ "%s: [verifier.environment] names no docker_image — grading in the agent's "
612
+ "image rather than building tests/Dockerfile, so the verifier runs somewhere "
613
+ "the task never declared",
614
+ task_dir.name,
615
+ )
616
+ unsupported = [
617
+ field
618
+ for field in ("healthcheck", "mcp_servers", "skills_dir", "gpu_types", "tpu")
619
+ if getattr(environment, field, None)
620
+ ]
621
+ if environment.os != TaskOS.LINUX or unsupported:
622
+ raise ValueError(
623
+ f"{task_dir.name}: verifier environment declares "
624
+ f"{unsupported or environment.os}, which verifiers' verifier-runtime "
625
+ "integration cannot honor"
626
+ )
627
+
628
+ network = parsed.verifier.explicit_phase_policy() or environment.resolve_baseline()
629
+ return VerifierConfig(
630
+ image=environment.docker_image if declared else None,
631
+ # A declared environment states its own resources; what it leaves out is the
632
+ # run's default, not the agent task's. A fresh copy is the task's environment,
633
+ # so it keeps whatever the agent box resolved to.
634
+ resources=(
635
+ task_resources(environment, harbor_config.resource_multiplier)
636
+ if declared
637
+ else TaskResources()
638
+ ),
639
+ workdir=environment.workdir if declared else None,
640
+ fresh_copy=not declared,
641
+ network_allow=(
642
+ ["*"]
643
+ if network.network_mode == NetworkMode.PUBLIC
644
+ else list(network.allowed_hosts)
645
+ ),
646
+ )
451
647
 
452
648
 
453
649
  def verifier_env(task: HarborData) -> dict[str, str]:
verifiers/v1/trace.py CHANGED
@@ -319,10 +319,11 @@ class Trace(BaseModel, Generic[DataT, StateT, AgentConfigT]):
319
319
  calls: list[ModelCall] = Field(default_factory=list)
320
320
  """Every model call; automatically recorded at intercept time + linked into `nodes`."""
321
321
 
322
- rewards: dict[str, Reward] = Field(default_factory=dict)
323
- """Named, weighted rewards"""
324
- metrics: dict[str, float] = Field(default_factory=dict)
325
- """Unweighted, named metrics"""
322
+ rewards: dict[str, Reward | None] = Field(default_factory=dict)
323
+ """Named, weighted rewards; `None` means scoring didn't run (e.g. because of a
324
+ preceding error)."""
325
+ metrics: dict[str, float | None] = Field(default_factory=dict)
326
+ """Unweighted, named metrics; `None` as in `rewards`."""
326
327
  info: dict[str, Any] = Field(default_factory=dict)
327
328
  """Scratch space for task-specific metadata."""
328
329
  state: StateT = Field(default_factory=State, exclude=True)
@@ -346,7 +347,7 @@ class Trace(BaseModel, Generic[DataT, StateT, AgentConfigT]):
346
347
 
347
348
  @property
348
349
  def reward(self) -> float:
349
- return sum(r.value for r in self.rewards.values())
350
+ return sum(r.value for r in self.rewards.values() if r is not None)
350
351
 
351
352
  @property
352
353
  def has_error(self) -> bool:
@@ -2,7 +2,7 @@
2
2
 
3
3
  import asyncio
4
4
  import inspect
5
- from collections.abc import Callable
5
+ from collections.abc import Callable, Iterable
6
6
  from typing import Any, TypeVar, overload
7
7
 
8
8
  F = TypeVar("F", bound=Callable[..., Any])
@@ -38,6 +38,22 @@ async def invoke_all(
38
38
  return await asyncio.gather(*(invoke(fn, available) for fn in fns))
39
39
 
40
40
 
41
+ def seed(scores: dict[str, Any], names: Iterable[str]) -> None:
42
+ """Mark every expected scoring key as unscored (`None`) before invocation, so a
43
+ trace records which signals should have run even when scoring fails midway.
44
+ Overwrites: a re-scoring attempt resets its own names, so stale values from a
45
+ previous attempt never read as fresh."""
46
+ for name in names:
47
+ scores[name] = None
48
+
49
+
50
+ def unseed(scores: dict[str, Any], name: str) -> None:
51
+ """Drop a still-unscored seed — a handler returning keyed scores records under
52
+ its result's keys, not its function name."""
53
+ if scores.get(name) is None:
54
+ scores.pop(name, None)
55
+
56
+
41
57
  def mark(attr: str, **extra: Any) -> Callable[[F], F]:
42
58
  def decorator(f: F) -> F:
43
59
  setattr(f, attr, True)
@@ -93,7 +93,8 @@ def trace_to_sample(
93
93
  # Flatten sub-rewards to top-level keys the way v0 does (raw scores, as v0's
94
94
  # per-function outputs were); env metrics stay nested.
95
95
  for name, reward in trace.rewards.items():
96
- sample.setdefault(name, reward.score)
96
+ if reward is not None:
97
+ sample.setdefault(name, reward.score)
97
98
  return sample
98
99
 
99
100
 
@@ -128,8 +129,15 @@ def _run_metrics(episodes: list[Episode], traces: list[Trace]) -> dict[str, Any]
128
129
  sums: dict[str, float] = {}
129
130
  counts: dict[str, int] = {}
130
131
  for trace in scored:
131
- scores = {name: reward.score for name, reward in trace.rewards.items()}
132
- for name, value in {**scores, **trace.metrics}.items():
132
+ scores = {
133
+ name: reward.score
134
+ for name, reward in trace.rewards.items()
135
+ if reward is not None
136
+ }
137
+ metrics = {
138
+ name: value for name, value in trace.metrics.items() if value is not None
139
+ }
140
+ for name, value in {**scores, **metrics}.items():
133
141
  sums[name] = sums.get(name, 0.0) + value
134
142
  counts[name] = counts.get(name, 0) + 1
135
143
  n = len(scored)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: verifiers
3
- Version: 0.2.2.dev73
3
+ Version: 0.2.2.dev75
4
4
  Summary: Verifiers: Environments for LLM Reinforcement Learning
5
5
  Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
6
6
  Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
@@ -168,20 +168,20 @@ verifiers/utils/tool_utils.py,sha256=gInWZODWQUZN4TyEMuuITyOUm9qlHJxtcvr1zZl_zJs
168
168
  verifiers/utils/usage_utils.py,sha256=Mw6WC78r6EsLyRFdP9ICveqptkjizbSkAC86ZH0geVU,3918
169
169
  verifiers/utils/version_utils.py,sha256=39gV8-mNF9GspAgKloLo8oy7_MVluMrwGPrntQv2hx4,2656
170
170
  verifiers/v1/__init__.py,sha256=kacTRsGsRHfOruv16ZyIOesfvCH_nW98ARKPYSgcNIc,7509
171
- verifiers/v1/agent.py,sha256=3kQNASeO0aedCN8m95RAzxu2TbK0rfG7Ix6-f2yxQ7Q,31545
171
+ verifiers/v1/agent.py,sha256=MQVKGjUhrP8uzRXAq94uJHkEsdHFziPxQHZ7TOD324I,31306
172
172
  verifiers/v1/env.py,sha256=g6fpoG-Z9lM3vTKrrU-QRJY3S88_y9zsPHwBvfs5vdY,17865
173
173
  verifiers/v1/episode.py,sha256=YtuRTV_PpwDglXFeN6UNg8IswpP1HseFLCYtascGMNI,3218
174
174
  verifiers/v1/errors.py,sha256=kQeEPX06TwAIuiz7MlWFYirrpKbx5STi6A5b_dtQQoA,6885
175
175
  verifiers/v1/graph.py,sha256=aBAzh1Ibp3klKNYjvzNyJSbjV8Oi5TJ791dN7o6ex1E,29070
176
- verifiers/v1/harness.py,sha256=i3pYHmdVPD7RtfZ-MAHlctaoLO8FxazCPmNy7mtPoSk,10954
176
+ verifiers/v1/harness.py,sha256=eKuqvYbQjnCMavkiYWZQDHJgH08m6SJ4PuevPuShjS0,11097
177
177
  verifiers/v1/judge.py,sha256=Pvr0C41ah1qkNSZ0WXdkDvBMfv5YSfY1REHncwlJ8cg,9346
178
178
  verifiers/v1/legacy.py,sha256=eLsFiwihOMh4zS_u1V7DXF22r9beoG4k2JGwNEcSPfg,22872
179
179
  verifiers/v1/rollout.py,sha256=s1f1VZAAAI89j3y9ceq_E4mvX0F7cFQAnyx27SZsjvA,17881
180
180
  verifiers/v1/session.py,sha256=tNAraQfiJKpWqgBFNLyPhfHiw0JhzzkyOfRcKWGnR2Q,6939
181
181
  verifiers/v1/state.py,sha256=EckF2bWp-vV4b1jYJ9sLI5xrfGuI5spIgYYwW926toI,595
182
- verifiers/v1/task.py,sha256=D5PKxu6WlqCjv7hE311aaNvBA3QWVWDuPIEwDsq-J7g,9075
182
+ verifiers/v1/task.py,sha256=jwMKiKlMtksTd8j2dcDlFtb6LC8BRE-N5jkNXlMp2jc,9457
183
183
  verifiers/v1/taskset.py,sha256=fp2E0IEhL_Ybj9cegZwljfTmW27p_30zTWHFKw_kXE4,4374
184
- verifiers/v1/trace.py,sha256=_bGHOrE6fRLGm0JYc0PrrrO3lyr-qqOF5Y1bHfViwR0,19347
184
+ verifiers/v1/trace.py,sha256=Z4uK-lYAbl-6CHFhTo-gadHnbi63LC0sJtUERC_m5ss,19477
185
185
  verifiers/v1/types.py,sha256=1PxamJspmoTc3OlFZAwH6o_2i_Tg9ZNvHxttIHWIxq8,8948
186
186
  verifiers/v1/acp/__init__.py,sha256=9SYCFtzGUM_wH98ldpVLcFhoq2M999jbG0tU5ODY70U,2297
187
187
  verifiers/v1/acp/_runner.py,sha256=BcYaNZHuawzCgxOVdhiF6PY_B1HMxIA1MwHy0mOzhSk,7740
@@ -196,7 +196,7 @@ verifiers/v1/cli/serve.py,sha256=_mdJaHVVe4pXzcpAot0ugcmiN3ra2gS4GrGmLghnX5M,286
196
196
  verifiers/v1/cli/validate.py,sha256=7ax6FqIzNBSYjd3JXK4v7Vbq34Rozyh1OvGpYnGqlSg,10266
197
197
  verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
198
198
  verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
199
- verifiers/v1/cli/dashboard/eval.py,sha256=nQ24ptQ_6ylmrB9VoZkgga1CLp_7sXZueC_MC4O9Dow,33452
199
+ verifiers/v1/cli/dashboard/eval.py,sha256=4j2YCPyzNHDT7d8MwqYR3C906lzwDVAd3FxgFYFgsw0,33588
200
200
  verifiers/v1/cli/dashboard/replay.py,sha256=CvRVaf0dUum5v0IzPQbFPPikBQmTrkMeaunGfVMTWHc,2755
201
201
  verifiers/v1/cli/dashboard/validate.py,sha256=rLsQ_31DjTIZpC_VO7zSG41ncUEBxqCx8mL9nTnGiMc,3650
202
202
  verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
@@ -234,7 +234,7 @@ verifiers/v1/dialects/chat.py,sha256=R_otqi5N0snB1QVKIYloNeLHR9kc7OmI06HYg1ir_Lw
234
234
  verifiers/v1/dialects/responses.py,sha256=ismnxUikPcPUIm80FUTPBbj-Cn1e0yGAsISMEseAWRc,13679
235
235
  verifiers/v1/envs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
236
236
  verifiers/v1/envs/agentic_judge/__init__.py,sha256=6vwtMCQ_cuvuubCF5-nrE5W7gGgzE-LIPKorkM5DXdw,309
237
- verifiers/v1/envs/agentic_judge/env.py,sha256=jreQPnbp8pJQSyZW0Me2IW-JipYsczqgohfM4-JCz9c,16968
237
+ verifiers/v1/envs/agentic_judge/env.py,sha256=6QrfuIlg-9MkcqIn5_6F3qJ6ddLJNaeYsZRJC9EzGkw,17011
238
238
  verifiers/v1/envs/best_of_n/__init__.py,sha256=mubASiwXMNIhnQFDccNAc7yZ0OFsOVK8hH8WIx9b2A4,119
239
239
  verifiers/v1/envs/best_of_n/env.py,sha256=KEUJBCVqblnXP_lEwm3mjGdaikBMkecz8UjOFPBhxc0,1987
240
240
  verifiers/v1/envs/single_agent/__init__.py,sha256=r0edAc_6gtHBhPzeBGVcdHZZqSdCPZnLMje6eVgyz5U,138
@@ -297,13 +297,13 @@ verifiers/v1/mcp/__init__.py,sha256=pm8ZWBiL1q1tY4wHvJzM_QrZ9oR8aGm-bIRWiRJsgHI,
297
297
  verifiers/v1/mcp/launch.py,sha256=wBpyeiOO86l1H2lVMbLVSV9Lbp90XeyXqauuK8snfEU,20171
298
298
  verifiers/v1/mcp/server.py,sha256=dRBxytFgS6rv1BS6t5EgBd3xeF6U1nTyt0ArDCLk3VE,12301
299
299
  verifiers/v1/mcp/toolset.py,sha256=T8Ioiq4wphxVOTHUDf28ZoQBM2ZYEevGnGgw8TFEMiA,1002
300
- verifiers/v1/runtimes/__init__.py,sha256=TqNtJVahzQuLS4fgCVeT9xdOzTkp0hRAPOkl7eFVDUg,2011
301
- verifiers/v1/runtimes/base.py,sha256=sTCiLo51yAU_T-Voc3BTH-FuzyEM0cGY05VzFE4DmYc,14720
300
+ verifiers/v1/runtimes/__init__.py,sha256=Kwe612RjtGzrQhc19QdXc7CmZAnQLH-EdK_xqkKP3Tk,2629
301
+ verifiers/v1/runtimes/base.py,sha256=tZRELSfZmi3l_1BsjabnNC3mEPfqNvUNevdnRvLPv6w,16257
302
302
  verifiers/v1/runtimes/limiters.py,sha256=ZjjYUFEt20yH8ggL5uU9w9U500AOB44qSUTQ7bYyh0c,2730
303
- verifiers/v1/runtimes/modal.py,sha256=FnocoJd9lQVQS3V3CoBon7KpKXm1bKWWeU2FAkq2bKU,8825
304
- verifiers/v1/runtimes/prime.py,sha256=zMMoB85uqwXjRTTyb3q5C7Xs4VmHagB8oI92Kuc-E1E,14755
305
- verifiers/v1/runtimes/subprocess.py,sha256=cs985D8-n_DOk88IRTyP1vfVlyUV1NBvS450lC-vw_s,5987
306
- verifiers/v1/runtimes/docker/__init__.py,sha256=eC5hjfyPwvwEKaeZUeP9EN8tJkrl0HUxGMzjiK4tK4Q,14555
303
+ verifiers/v1/runtimes/modal.py,sha256=eFiT-Ovf_RwQZbSw3Dwv2EXeHV0d6uGvPswqKYKCw5k,8826
304
+ verifiers/v1/runtimes/prime.py,sha256=kwvPqUI8D-GvcFwYRdpTHE1hXS0vfJGPlFOLgzm4Mvs,14756
305
+ verifiers/v1/runtimes/subprocess.py,sha256=ZrHRlTuTyCrcSKpiPD2ElcPpe97EvVsXJ8BfwUaNXiI,5988
306
+ verifiers/v1/runtimes/docker/__init__.py,sha256=vyEgf6XgAwjGgBicJchJvNgcDgYXrRvyQC4W52-23GU,14556
307
307
  verifiers/v1/runtimes/docker/egress.py,sha256=37UXgNX9gnoLxjMg4KZaJIGoApCAM6MK22eJGNoMi38,14514
308
308
  verifiers/v1/serve/__init__.py,sha256=sr_zkDzAHnOcQm-QRm8xwJBmCMFD-xhKIFPaDZvMbUs,641
309
309
  verifiers/v1/serve/client.py,sha256=CxpOwCn4ehuCXfqDJeV9zj7O0XC7MPcI-OZ1L_CSDlo,7057
@@ -311,8 +311,9 @@ verifiers/v1/serve/pool.py,sha256=AyOMjDYi9PLGA8Z-bnr6Fd5cnubuRavEBuudbVWtq2E,14
311
311
  verifiers/v1/serve/server.py,sha256=eFWJsTUDbMywKQXKDuk57QvS7Y4IZwW2Ok3LwJHMFLI,9200
312
312
  verifiers/v1/serve/types.py,sha256=3F4IIGIQgYITlEFxeX3uNJYrJlKugFrPTozbDNwDfBY,2674
313
313
  verifiers/v1/tasksets/__init__.py,sha256=b62O4WDNbGfJH_4NjYmznn5otWovdfvUgJZyoIZyLh8,596
314
- verifiers/v1/tasksets/harbor/__init__.py,sha256=wwVGA8N7E2Q2qA5WGHuLRrn4DgG5oaRHrCS2RW3NXVY,195
315
- verifiers/v1/tasksets/harbor/taskset.py,sha256=9rfWTg0s3zPAEppZnjh7zNyi2EbM-qORK2lyWQZQL7s,19565
314
+ verifiers/v1/tasksets/harbor/__init__.py,sha256=JxThzHFYEkuSfQbVf3pEnjordgdfe4MU-QOsSSbyovU,326
315
+ verifiers/v1/tasksets/harbor/env.py,sha256=bNF5dfbhcsdrd_8JmW5ved8Yk-mHhrOgKkVEA563qEo,5847
316
+ verifiers/v1/tasksets/harbor/taskset.py,sha256=vmDjP7dhxchZ5-J9IUeiATwwchxUvAeciCevD6W_N4c,29032
316
317
  verifiers/v1/tasksets/lean/__init__.py,sha256=oyv-GbJQBvCyGwo7Ysxm2NjZW-aX4j1uDoHU0f-E7xQ,767
317
318
  verifiers/v1/tasksets/lean/scoring.py,sha256=sfzsT6MUK0zdMgAQ57_P_0CBaPHy3M7go71QMr5hDrs,10599
318
319
  verifiers/v1/tasksets/lean/taskset.py,sha256=XgUqX0uJvHC8sKFhbK92iu6nUGcPuXCUpI3ZBhDcb7c,9025
@@ -324,7 +325,7 @@ verifiers/v1/utils/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuF
324
325
  verifiers/v1/utils/aio.py,sha256=0yNFOqB2oO6Qktc6NI_ZSAGuo1aHsIK_wuT1BtmoZ3I,1532
325
326
  verifiers/v1/utils/artifacts.py,sha256=vOzbPSuMirUecyylrhvLZ_vAcTd6JTJBztGPDQ0ANNk,6706
326
327
  verifiers/v1/utils/compile.py,sha256=Oh4hDVRNsql_BSOMh4kDEuze9_ZF17Q7uzBhhU70UZA,5663
327
- verifiers/v1/utils/decorators.py,sha256=9CcCwgZlRjVGPqT5G1CFJUHP4OIMuBGkUJZkcI6dnbk,3865
328
+ verifiers/v1/utils/decorators.py,sha256=DG_K3HMiWqRHsY1bA7FJD3uTHUVuLw6SUJV-PnNRifA,4538
328
329
  verifiers/v1/utils/format.py,sha256=NfQo9M5KMzWCZpQaet5YtsJx56vCKAZPfbjZZJLJhhM,2187
329
330
  verifiers/v1/utils/generic.py,sha256=2VIN9RapMO9pVkFHB1UKUHck1vlTJ1pjfad2NzxuuXA,2055
330
331
  verifiers/v1/utils/git.py,sha256=MTRinOvG4EddfjdahHkFZUihHmhJ37n5ROtgjzjE1NA,8366
@@ -334,12 +335,12 @@ verifiers/v1/utils/interrupt.py,sha256=F-KKhc5ndPJJfhd3SuMqyqhXhA32FhCRy5KWFJpEo
334
335
  verifiers/v1/utils/loaders.py,sha256=NSIBdNU3JVeyddL-dM4dJt6khCHyjCgzJuFtLrgiwD0,9989
335
336
  verifiers/v1/utils/logging.py,sha256=OcMHA6NsYux3oIzjPuI95rDWmFBHNcHDjZeNIhXTX-Y,2084
336
337
  verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
337
- verifiers/v1/utils/platform.py,sha256=MWZmvyj0iyzNIO0xRhlkCXtFF1plq6EJi8rWlvrFtvI,11363
338
+ verifiers/v1/utils/platform.py,sha256=56Ixmk1SER6q5LvzyYcA0hgzGpzhxKZL8Hiqp4c_XVc,11576
338
339
  verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
339
340
  verifiers/v1/utils/score.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
340
341
  verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
341
- verifiers-0.2.2.dev73.dist-info/METADATA,sha256=DU0GJRGJc9sAtCc2vIZa5E-O-4ubaCtIXJiCZQHs2rc,4545
342
- verifiers-0.2.2.dev73.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
343
- verifiers-0.2.2.dev73.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
344
- verifiers-0.2.2.dev73.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
345
- verifiers-0.2.2.dev73.dist-info/RECORD,,
342
+ verifiers-0.2.2.dev75.dist-info/METADATA,sha256=8ZNGDY1kLfAvLaXC5voMOQeGRtvHeflfC3eV0v34rWg,4545
343
+ verifiers-0.2.2.dev75.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
344
+ verifiers-0.2.2.dev75.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
345
+ verifiers-0.2.2.dev75.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
346
+ verifiers-0.2.2.dev75.dist-info/RECORD,,