verifiers 0.3.2.dev99__py3-none-any.whl → 0.3.2.dev101__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/agent.py +8 -0
- verifiers/v1/cli/dashboard/eval.py +14 -9
- verifiers/v1/cli/debug.py +1 -1
- verifiers/v1/cli/eval/hint.py +5 -0
- verifiers/v1/cli/eval/main.py +5 -7
- verifiers/v1/cli/eval/runner.py +20 -129
- verifiers/v1/cli/gepa.py +2 -2
- verifiers/v1/cli/init.py +3 -3
- verifiers/v1/cli/replay.py +1 -1
- verifiers/v1/cli/validate.py +2 -2
- verifiers/v1/configs/agent.py +2 -1
- verifiers/v1/configs/cli/eval.py +3 -10
- verifiers/v1/configs/cli/validate.py +1 -1
- verifiers/v1/configs/client.py +1 -1
- verifiers/v1/env.py +9 -2
- verifiers/v1/graph.py +6 -0
- verifiers/v1/harnesses/rlm/harness.py +17 -10
- verifiers/v1/interception/server.py +8 -0
- verifiers/v1/rollout.py +6 -0
- verifiers/v1/serve/__init__.py +2 -0
- verifiers/v1/serve/client.py +75 -34
- verifiers/v1/serve/delta.py +281 -0
- verifiers/v1/serve/pool.py +19 -10
- verifiers/v1/serve/server.py +29 -8
- verifiers/v1/serve/types.py +12 -6
- verifiers/v1/tasksets/harbor/taskset.py +4 -4
- verifiers/v1/trace.py +29 -0
- {verifiers-0.3.2.dev99.dist-info → verifiers-0.3.2.dev101.dist-info}/METADATA +1 -1
- {verifiers-0.3.2.dev99.dist-info → verifiers-0.3.2.dev101.dist-info}/RECORD +32 -30
- verifiers-0.3.2.dev101.dist-info/entry_points.txt +7 -0
- verifiers-0.3.2.dev99.dist-info/entry_points.txt +0 -7
- {verifiers-0.3.2.dev99.dist-info → verifiers-0.3.2.dev101.dist-info}/WHEEL +0 -0
- {verifiers-0.3.2.dev99.dist-info → verifiers-0.3.2.dev101.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/agent.py
CHANGED
|
@@ -54,11 +54,19 @@ __all__ = ["Agent", "AgentConfig", "Agents", "TimeoutConfig", "make_agent"]
|
|
|
54
54
|
logger = logging.getLogger(__name__)
|
|
55
55
|
|
|
56
56
|
|
|
57
|
+
DEFAULT_ROLLOUT_TIMEOUT = 4 * 3600.0
|
|
58
|
+
"""Agent solve-attempt budget when neither the eval config nor the task sets one."""
|
|
59
|
+
|
|
60
|
+
|
|
57
61
|
def resolve_rollout_timeouts(timeout: TimeoutConfig, task: Task) -> RolloutTimeouts:
|
|
58
62
|
"""Apply an agent's stage-timeout precedence to one task."""
|
|
59
63
|
agent_timeout = (
|
|
60
64
|
timeout.rollout if timeout.rollout is not None else task.data.timeout.agent
|
|
61
65
|
)
|
|
66
|
+
if agent_timeout is None:
|
|
67
|
+
agent_timeout = DEFAULT_ROLLOUT_TIMEOUT
|
|
68
|
+
elif agent_timeout == 0:
|
|
69
|
+
agent_timeout = None # explicit: unbounded
|
|
62
70
|
return RolloutTimeouts(
|
|
63
71
|
setup=timeout.setup if timeout.setup is not None else task.data.timeout.setup,
|
|
64
72
|
agent=agent_timeout,
|
|
@@ -16,6 +16,7 @@ from rich.table import Table
|
|
|
16
16
|
from rich.text import Text
|
|
17
17
|
|
|
18
18
|
from verifiers.v1.cli.dashboard.base import live_view
|
|
19
|
+
from verifiers.v1.cli.eval.hint import PRIME_RL_HINT
|
|
19
20
|
from verifiers.v1.cli.output import attempt_log_file, output_path
|
|
20
21
|
from verifiers.v1.configs.cli.eval import EvalConfig
|
|
21
22
|
from verifiers.v1.env import RunSlot
|
|
@@ -280,6 +281,10 @@ def Overview(config: EvalConfig) -> Table:
|
|
|
280
281
|
return grid
|
|
281
282
|
|
|
282
283
|
|
|
284
|
+
def _prime_rl_footer() -> Group:
|
|
285
|
+
return Group(Text(""), Text(PRIME_RL_HINT, style="dim", overflow="fold"))
|
|
286
|
+
|
|
287
|
+
|
|
283
288
|
def _push_footer(push: "PushState | None") -> Group | None:
|
|
284
289
|
"""The `--push` line under the rollouts: dim with the run's URL while it streams,
|
|
285
290
|
white once pushed, red when it failed. `None` when `--push` is off or the run stayed
|
|
@@ -833,13 +838,15 @@ def _render(
|
|
|
833
838
|
# The --push status line (and, on Ctrl-C, the cleanup notice) appear under the rollouts. Measure
|
|
834
839
|
# the fixed top (header + progress + rule) and the footer so the rollout rows fill what's left;
|
|
835
840
|
# page through them (timer / arrows) when they'd overflow (else rich truncates).
|
|
836
|
-
footers = [
|
|
837
|
-
|
|
841
|
+
footers = [
|
|
842
|
+
f
|
|
843
|
+
for f in (_push_footer(push), _interrupt_footer(), _prime_rl_footer())
|
|
844
|
+
if f is not None
|
|
845
|
+
]
|
|
846
|
+
footer = Group(*footers)
|
|
838
847
|
progress = Progress(slots, start, completed)
|
|
839
848
|
top = Group(header, progress, Rule(style="dim"))
|
|
840
|
-
reserved = len(_CONSOLE.render_lines(top))
|
|
841
|
-
if footer is not None:
|
|
842
|
-
reserved += len(_CONSOLE.render_lines(footer))
|
|
849
|
+
reserved = len(_CONSOLE.render_lines(top)) + len(_CONSOLE.render_lines(footer))
|
|
843
850
|
rows_per_page = max(1, _CONSOLE.size.height - reserved - 1)
|
|
844
851
|
if tail is not None: # --show-logs: the run's log stream in place of rollout rows
|
|
845
852
|
parts = [
|
|
@@ -847,9 +854,8 @@ def _render(
|
|
|
847
854
|
progress,
|
|
848
855
|
Rule(style="dim"),
|
|
849
856
|
tail.view(rows_per_page),
|
|
857
|
+
footer,
|
|
850
858
|
]
|
|
851
|
-
if footer is not None:
|
|
852
|
-
parts.append(footer)
|
|
853
859
|
return Group(*parts)
|
|
854
860
|
page_groups, index, count = _paginate(_groups(slots), rows_per_page, pager, now)
|
|
855
861
|
if count > 1:
|
|
@@ -859,9 +865,8 @@ def _render(
|
|
|
859
865
|
progress,
|
|
860
866
|
Rule(style="dim"),
|
|
861
867
|
Rows(page_groups, now, runtime_type, completed),
|
|
868
|
+
footer,
|
|
862
869
|
]
|
|
863
|
-
if footer is not None:
|
|
864
|
-
parts.append(footer)
|
|
865
870
|
return Group(*parts)
|
|
866
871
|
|
|
867
872
|
|
verifiers/v1/cli/debug.py
CHANGED
|
@@ -36,7 +36,7 @@ from verifiers.v1.utils.logging import setup_logging
|
|
|
36
36
|
logger = logging.getLogger(__name__)
|
|
37
37
|
|
|
38
38
|
USAGE = (
|
|
39
|
-
"usage: uv run debug [<taskset-id>] (--command <cmd> | --script-path <path>) "
|
|
39
|
+
"usage: uv run vf-debug [<taskset-id>] (--command <cmd> | --script-path <path>) "
|
|
40
40
|
"[--runtime.type subprocess] [options] [@ file.toml]\n"
|
|
41
41
|
" runs setup, then one command or uploaded host script, and saves traces"
|
|
42
42
|
)
|
verifiers/v1/cli/eval/main.py
CHANGED
|
@@ -30,8 +30,8 @@ from verifiers.v1.utils.logging import setup_logging
|
|
|
30
30
|
logger = logging.getLogger(__name__)
|
|
31
31
|
|
|
32
32
|
USAGE = (
|
|
33
|
-
"usage: uv run eval [<taskset-id>] [--env.id <id>] [options] [@ file.toml]\n"
|
|
34
|
-
" uv run eval @ <run-dir>/configs/resolved/eval.json --resume (re-run the run's missing/errored rollouts)"
|
|
33
|
+
"usage: uv run vf-eval [<taskset-id>] [--env.id <id>] [options] [@ file.toml]\n"
|
|
34
|
+
" uv run vf-eval @ <run-dir>/configs/resolved/eval.json --resume (re-run the run's missing/errored rollouts)"
|
|
35
35
|
)
|
|
36
36
|
|
|
37
37
|
|
|
@@ -48,7 +48,7 @@ def main(argv: list[str] | None = None) -> None:
|
|
|
48
48
|
return
|
|
49
49
|
# An env-block flag skips the usage gate so the typed parse renders its
|
|
50
50
|
# did-you-mean instead of a bare usage line.
|
|
51
|
-
typed_axis = any(a.startswith(
|
|
51
|
+
typed_axis = any(a.startswith("--env.") for a in argv)
|
|
52
52
|
if (
|
|
53
53
|
not extract_id(argv, "env.taskset")
|
|
54
54
|
and not references_config_file(argv)
|
|
@@ -108,7 +108,7 @@ def main(argv: list[str] | None = None) -> None:
|
|
|
108
108
|
raise SystemExit(
|
|
109
109
|
f"--resume requires the exact config the run was started with - it "
|
|
110
110
|
f"differs in [{', '.join(changed)}]. Resumed rollouts would not be "
|
|
111
|
-
f"comparable; re-run with `uv run eval @ {saved_path} --resume`, or "
|
|
111
|
+
f"comparable; re-run with `uv run vf-eval @ {saved_path} --resume`, or "
|
|
112
112
|
"start a fresh run"
|
|
113
113
|
)
|
|
114
114
|
if config.dry_run: # resolved + validated; write it to the output dir and exit
|
|
@@ -116,8 +116,7 @@ def main(argv: list[str] | None = None) -> None:
|
|
|
116
116
|
logger.info("wrote config to %s", write_config(config, run_path))
|
|
117
117
|
return
|
|
118
118
|
# Always tee this attempt's logs to `logs/attempt_<n>/eval.log` (`logs/latest`
|
|
119
|
-
# points there)
|
|
120
|
-
# `--rich.show-logs` tails it live.
|
|
119
|
+
# points there); `--rich.show-logs` tails it live.
|
|
121
120
|
log_file = str(create_attempt_log_dir(run_path) / "eval.log")
|
|
122
121
|
level = "DEBUG" if config.verbose else "INFO"
|
|
123
122
|
setup_logging(level, log_file=log_file, console=config.rich is None)
|
|
@@ -128,7 +127,6 @@ def main(argv: list[str] | None = None) -> None:
|
|
|
128
127
|
install_interrupt()
|
|
129
128
|
|
|
130
129
|
try:
|
|
131
|
-
# Through the env-server worker pool by default; in-process with --no-serve.
|
|
132
130
|
episodes = asyncio.run(run_eval(config))
|
|
133
131
|
except KeyboardInterrupt:
|
|
134
132
|
# Graceful cleanup has already run (each rollout's `finally`); partial results are on
|
verifiers/v1/cli/eval/runner.py
CHANGED
|
@@ -1,12 +1,9 @@
|
|
|
1
1
|
"""The eval runner: fan episodes out with bounded concurrency.
|
|
2
2
|
|
|
3
|
-
Rollouts run
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
becomes one episode: `env.run_slot` in-process, a `run` request to the pool
|
|
8
|
-
otherwise. The dashboard watches the same `RunSlot`s either way; a served slot
|
|
9
|
-
has no live traces, so its per-turn detail lands when the episode completes.
|
|
3
|
+
Rollouts run in-process: the env comes up once in this process and each slot becomes
|
|
4
|
+
one episode through `env.run_slot`. The dashboard watches the same `RunSlot`s the env
|
|
5
|
+
fills with live traces. Serving an env to many consumers is prime-rl's job (its `eval`
|
|
6
|
+
runs env servers); this CLI is the quick local path.
|
|
10
7
|
"""
|
|
11
8
|
|
|
12
9
|
import asyncio
|
|
@@ -18,16 +15,15 @@ from typing import TypeVar, cast
|
|
|
18
15
|
|
|
19
16
|
from verifiers.v1.cli.dashboard import dashboard
|
|
20
17
|
from verifiers.v1.cli.eval import resume
|
|
18
|
+
from verifiers.v1.cli.eval.hint import PRIME_RL_HINT
|
|
21
19
|
from verifiers.v1.cli.output import (
|
|
22
20
|
append_episode,
|
|
23
|
-
attempt_log_file,
|
|
24
21
|
output_path,
|
|
25
22
|
save_config,
|
|
26
23
|
)
|
|
27
24
|
from verifiers.v1.cli.resume import distribute
|
|
28
25
|
from verifiers.v1.clients import ModelContext
|
|
29
26
|
from verifiers.v1.configs.cli.eval import EvalConfig
|
|
30
|
-
from verifiers.v1.configs.serve import ServeConfig
|
|
31
27
|
from verifiers.v1.env import Env, RunSlot
|
|
32
28
|
from verifiers.v1.episode import Episode, EvalRunInfo
|
|
33
29
|
from verifiers.v1.utils.aio import run_shielded
|
|
@@ -84,93 +80,11 @@ async def _in_process(
|
|
|
84
80
|
yield run
|
|
85
81
|
|
|
86
82
|
|
|
87
|
-
@contextlib.asynccontextmanager
|
|
88
|
-
async def _server(
|
|
89
|
-
config: EvalConfig,
|
|
90
|
-
serve: ServeConfig,
|
|
91
|
-
semaphore: asyncio.Semaphore | None,
|
|
92
|
-
on_complete: OnComplete,
|
|
93
|
-
) -> AsyncIterator[RunSlotFn]:
|
|
94
|
-
"""Run slots through a spawned env-server worker pool: each rollout is its own
|
|
95
|
-
`run` request, dispatched least-busy across workers. The workers own the env
|
|
96
|
-
(and its serving resources); this process owns the taskset and the results."""
|
|
97
|
-
import multiprocessing as mp
|
|
98
|
-
from functools import partial
|
|
99
|
-
|
|
100
|
-
from verifiers.v1.configs.serve import pool_serve_kwargs
|
|
101
|
-
from verifiers.v1.serve import EnvClient, env_config_data, serve_env
|
|
102
|
-
from verifiers.v1.utils.logging import setup_logging
|
|
103
|
-
|
|
104
|
-
# Spawned processes inherit no logging — hand them the main process's setup so
|
|
105
|
-
# their rollout logs land in the output dir. They share its stderr, so console
|
|
106
|
-
# output follows the main process's choice: off under the dashboard (worker log
|
|
107
|
-
# lines would print over the Live view and shift it), on otherwise.
|
|
108
|
-
level = "DEBUG" if config.verbose else "INFO"
|
|
109
|
-
log_file = str(attempt_log_file(output_path(config)))
|
|
110
|
-
console = config.rich is None
|
|
111
|
-
mpctx = mp.get_context("spawn")
|
|
112
|
-
address_queue: mp.Queue = mpctx.Queue()
|
|
113
|
-
# Death pipe: serve_env self-terminates if this process dies abruptly — we keep
|
|
114
|
-
# parent_conn, whose close (even on our SIGKILL) signals the child's watch.
|
|
115
|
-
parent_conn, child_conn = mpctx.Pipe()
|
|
116
|
-
proc = mpctx.Process(
|
|
117
|
-
target=serve_env,
|
|
118
|
-
kwargs=dict(
|
|
119
|
-
**pool_serve_kwargs(serve.pool),
|
|
120
|
-
address="tcp://127.0.0.1:0",
|
|
121
|
-
address_queue=address_queue,
|
|
122
|
-
death_pipe=child_conn,
|
|
123
|
-
log_setup=partial(setup_logging, level, log_file, console),
|
|
124
|
-
config_data=env_config_data(config.env), # picklable across the spawn
|
|
125
|
-
# `-c` seeds each worker's episode bound unless `[serve]` pins one — so a
|
|
126
|
-
# pool carries `workers * bound` episodes, as `multiplex` implies.
|
|
127
|
-
max_concurrent=serve.max_concurrent
|
|
128
|
-
if serve.max_concurrent is not None
|
|
129
|
-
else config.max_concurrent,
|
|
130
|
-
),
|
|
131
|
-
daemon=False,
|
|
132
|
-
)
|
|
133
|
-
proc.start()
|
|
134
|
-
child_conn.close() # the child holds its end; we keep parent_conn so our exit closes it
|
|
135
|
-
try:
|
|
136
|
-
address = await asyncio.to_thread(address_queue.get, timeout=600)
|
|
137
|
-
client = EnvClient(address=address)
|
|
138
|
-
try:
|
|
139
|
-
await client.wait_for_server_startup(timeout=600)
|
|
140
|
-
|
|
141
|
-
async def run(slot: RunSlot) -> Episode:
|
|
142
|
-
async with semaphore or contextlib.nullcontext():
|
|
143
|
-
slot.started = time.time()
|
|
144
|
-
episode = await client.run(
|
|
145
|
-
client=config.client,
|
|
146
|
-
model=config.model,
|
|
147
|
-
sampling=config.sampling,
|
|
148
|
-
task_data=slot.task.data.model_dump(mode="json"),
|
|
149
|
-
)
|
|
150
|
-
slot.traces = list(episode.traces)
|
|
151
|
-
slot.episode = cast(Episode, episode)
|
|
152
|
-
slot.done = True
|
|
153
|
-
await on_complete(cast(Episode, episode))
|
|
154
|
-
return cast(Episode, episode)
|
|
155
|
-
|
|
156
|
-
yield run
|
|
157
|
-
finally:
|
|
158
|
-
await client.close()
|
|
159
|
-
finally:
|
|
160
|
-
proc.terminate()
|
|
161
|
-
with contextlib.suppress(Exception):
|
|
162
|
-
await asyncio.to_thread(proc.join, 10)
|
|
163
|
-
with contextlib.suppress(Exception):
|
|
164
|
-
parent_conn.close()
|
|
165
|
-
|
|
166
|
-
|
|
167
83
|
async def run_eval(config: EvalConfig) -> list[Episode]:
|
|
168
|
-
from verifiers.v1.utils.loaders import load_environment
|
|
84
|
+
from verifiers.v1.utils.loaders import load_environment
|
|
169
85
|
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
env = None if config.serve is not None else load_environment(config.env)
|
|
173
|
-
taskset = env.taskset if env is not None else load_taskset(config.env.taskset)
|
|
86
|
+
env = load_environment(config.env)
|
|
87
|
+
taskset = env.taskset
|
|
174
88
|
if config.num_tasks is None and taskset.INFINITE:
|
|
175
89
|
raise ValueError(
|
|
176
90
|
f"{type(taskset).__name__} is infinite - bound the run with -n"
|
|
@@ -186,14 +100,13 @@ async def run_eval(config: EvalConfig) -> list[Episode]:
|
|
|
186
100
|
finished: list[Episode] = []
|
|
187
101
|
if config.resume:
|
|
188
102
|
keys = [task.hash for task in tasks]
|
|
189
|
-
#
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
103
|
+
# the env's own keep-verdict decides what resumes
|
|
104
|
+
loaded, owed = resume.load(
|
|
105
|
+
out,
|
|
106
|
+
keys,
|
|
107
|
+
config.num_rollouts,
|
|
108
|
+
lambda episode: env.complete(cast(Episode, episode)),
|
|
195
109
|
)
|
|
196
|
-
loaded, owed = resume.load(out, keys, config.num_rollouts, complete)
|
|
197
110
|
finished = [cast(Episode, episode) for episode in loaded]
|
|
198
111
|
if not owed: # already complete - report it and exit successfully
|
|
199
112
|
print(
|
|
@@ -211,18 +124,10 @@ async def run_eval(config: EvalConfig) -> list[Episode]:
|
|
|
211
124
|
)
|
|
212
125
|
else:
|
|
213
126
|
save_config(config, out)
|
|
214
|
-
via = (
|
|
215
|
-
f" via the env-server {config.serve.pool.type} pool"
|
|
216
|
-
if config.serve is not None
|
|
217
|
-
else ""
|
|
218
|
-
)
|
|
219
127
|
logger.info(
|
|
220
|
-
"running %dx%d rollouts on %s
|
|
221
|
-
len(plan),
|
|
222
|
-
config.num_rollouts,
|
|
223
|
-
config.model,
|
|
224
|
-
via,
|
|
128
|
+
"running %dx%d rollouts on %s", len(plan), config.num_rollouts, config.model
|
|
225
129
|
)
|
|
130
|
+
logger.info(PRIME_RL_HINT)
|
|
226
131
|
start = time.time()
|
|
227
132
|
logger.info("results: %s", out)
|
|
228
133
|
|
|
@@ -242,25 +147,11 @@ async def run_eval(config: EvalConfig) -> list[Episode]:
|
|
|
242
147
|
await append_episode(out, episode, write_lock)
|
|
243
148
|
await asyncio.to_thread(log_episodes, run, [episode])
|
|
244
149
|
|
|
245
|
-
|
|
246
|
-
_in_process(env, config, semaphore, on_complete)
|
|
247
|
-
if env is not None
|
|
248
|
-
else _server(config, config.serve, semaphore, on_complete)
|
|
249
|
-
)
|
|
250
|
-
# The run is closed out whatever breaks, backend setup and teardown included.
|
|
150
|
+
# The run is closed out whatever breaks, env setup and teardown included.
|
|
251
151
|
try:
|
|
252
|
-
async with
|
|
253
|
-
#
|
|
254
|
-
|
|
255
|
-
planned = [
|
|
256
|
-
slot
|
|
257
|
-
for task, n in plan
|
|
258
|
-
for slot in (
|
|
259
|
-
env.slots(task, n)
|
|
260
|
-
if env is not None
|
|
261
|
-
else [RunSlot(task) for _ in range(n)]
|
|
262
|
-
)
|
|
263
|
-
]
|
|
152
|
+
async with _in_process(env, config, semaphore, on_complete) as run_slot:
|
|
153
|
+
# the env's own slots: it fills their live traces as the rollouts run
|
|
154
|
+
planned = [slot for task, n in plan for slot in env.slots(task, n)]
|
|
264
155
|
slots = [RunSlot.finished(episode) for episode in finished] + planned
|
|
265
156
|
display = (
|
|
266
157
|
dashboard(slots, config, start, push=push_state)
|
verifiers/v1/cli/gepa.py
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
"""The GEPA entrypoint: `uv run gepa [<taskset-id>] --model <model> [options]`.
|
|
1
|
+
"""The GEPA entrypoint: `uv run vf-gepa [<taskset-id>] --model <model> [options]`.
|
|
2
2
|
|
|
3
3
|
Registered as the `gepa` console script. Optimizes a v1 taskset's `Task.system_prompt` via
|
|
4
4
|
GEPA (Genetic-Pareto): alternating rollouts with a teacher LM reflecting on results — see
|
|
@@ -28,7 +28,7 @@ from verifiers.v1.utils.logging import setup_logging
|
|
|
28
28
|
|
|
29
29
|
logger = logging.getLogger(__name__)
|
|
30
30
|
|
|
31
|
-
USAGE = "usage: uv run gepa [<taskset-id>] [--env.id <id>] --model <model> [options] [@ file.toml]"
|
|
31
|
+
USAGE = "usage: uv run vf-gepa [<taskset-id>] [--env.id <id>] --model <model> [options] [@ file.toml]"
|
|
32
32
|
|
|
33
33
|
|
|
34
34
|
def main(argv: list[str] | None = None) -> None:
|
verifiers/v1/cli/init.py
CHANGED
|
@@ -8,7 +8,7 @@ from pydantic_config import cli
|
|
|
8
8
|
from verifiers.v1.configs.cli.init import InitConfig
|
|
9
9
|
|
|
10
10
|
USAGE = (
|
|
11
|
-
"usage: uv run init <name> [--path ./environments] [-T/--add-tool] "
|
|
11
|
+
"usage: uv run vf-init <name> [--path ./environments] [-T/--add-tool] "
|
|
12
12
|
"[-H/--add-harness]\n"
|
|
13
13
|
" scaffold a new environment package"
|
|
14
14
|
)
|
|
@@ -191,7 +191,7 @@ A v1 verifiers environment, scaffolded with `init`.
|
|
|
191
191
|
|
|
192
192
|
```bash
|
|
193
193
|
uv pip install -e . # install this package (or register it in your project)
|
|
194
|
-
uv run eval {dash} -n 3 # evaluate a few tasks with the bash harness
|
|
194
|
+
uv run vf-eval {dash} -n 3 # evaluate a few tasks with the bash harness
|
|
195
195
|
```
|
|
196
196
|
|
|
197
197
|
## Layout
|
|
@@ -228,7 +228,7 @@ def scaffold(config: InitConfig) -> Path:
|
|
|
228
228
|
_write(pkg_dir / "servers" / "__init__.py", "")
|
|
229
229
|
_write(pkg_dir / "servers" / "tool.py", _tool_py(stem, prefix))
|
|
230
230
|
|
|
231
|
-
print(f"\ndone. next:\n uv pip install -e {env_dir}\n uv run eval {dash} -n 3")
|
|
231
|
+
print(f"\ndone. next:\n uv pip install -e {env_dir}\n uv run vf-eval {dash} -n 3")
|
|
232
232
|
return env_dir
|
|
233
233
|
|
|
234
234
|
|
verifiers/v1/cli/replay.py
CHANGED
|
@@ -39,7 +39,7 @@ from verifiers.v1.utils.logging import setup_logging
|
|
|
39
39
|
logger = logging.getLogger(__name__)
|
|
40
40
|
|
|
41
41
|
USAGE = (
|
|
42
|
-
"usage: uv run replay <output-dir> [options] [@ file.toml]\n"
|
|
42
|
+
"usage: uv run vf-replay <output-dir> [options] [@ file.toml]\n"
|
|
43
43
|
" re-score a finished run's saved traces (judges + trace-only signals; no runtime)"
|
|
44
44
|
)
|
|
45
45
|
|
verifiers/v1/cli/validate.py
CHANGED
|
@@ -49,9 +49,9 @@ REASONS = ("valid", "invalid", "unchecked", "error", "timeout")
|
|
|
49
49
|
ResultRow = dict[str, Any]
|
|
50
50
|
|
|
51
51
|
USAGE = (
|
|
52
|
-
"usage: uv run validate [<taskset-id>] [--only-setup | --only-gold] "
|
|
52
|
+
"usage: uv run vf-validate [<taskset-id>] [--only-setup | --only-gold] "
|
|
53
53
|
"[-o <output-dir>] [--runtime.type subprocess] [options] [@ file.toml]\n"
|
|
54
|
-
" uv run validate @ <run-dir>/configs/validate.json --resume (re-run missing/errored/timed-out tasks)\n"
|
|
54
|
+
" uv run vf-validate @ <run-dir>/configs/validate.json --resume (re-run missing/errored/timed-out tasks)\n"
|
|
55
55
|
" runs persisted gold and setup-only checks per task (no model)"
|
|
56
56
|
)
|
|
57
57
|
|
verifiers/v1/configs/agent.py
CHANGED
|
@@ -16,7 +16,8 @@ class TimeoutConfig(BaseConfig):
|
|
|
16
16
|
setup: float | None = None # one shared budget: task setup + provisioning
|
|
17
17
|
"""Timeout (in seconds) for the task + harness setup hooks."""
|
|
18
18
|
rollout: float | None = None
|
|
19
|
-
"""Timeout (in seconds) for the agent's solve attempt.
|
|
19
|
+
"""Timeout (in seconds) for the agent's solve attempt. Unset: the task's own
|
|
20
|
+
timeout, else 4 hours. `0` disables the timeout."""
|
|
20
21
|
finalize: float | None = None
|
|
21
22
|
"""Timeout (in seconds) for the task + harness finalize hooks."""
|
|
22
23
|
scoring: float | None = None
|
verifiers/v1/configs/cli/eval.py
CHANGED
|
@@ -9,7 +9,6 @@ from pydantic_config import BaseConfig
|
|
|
9
9
|
from verifiers.v1.clients import ClientConfig, EvalClientConfig
|
|
10
10
|
from verifiers.v1.configs.cli.env import narrowed_env_annotation, resolve_env_field
|
|
11
11
|
from verifiers.v1.configs.env import EnvConfig
|
|
12
|
-
from verifiers.v1.configs.serve import ServeConfig
|
|
13
12
|
from verifiers.v1.envs.single_agent import SingleAgentEnvConfig
|
|
14
13
|
from verifiers.v1.types import SamplingConfig
|
|
15
14
|
|
|
@@ -76,10 +75,6 @@ class EvalConfig(BaseConfig):
|
|
|
76
75
|
env: SerializeAsAny[EnvConfig] = SingleAgentEnvConfig()
|
|
77
76
|
"""The environment — which env, its seed taskset, each agent, its knobs. Narrowed to
|
|
78
77
|
the selected env's config class by the env id, else the taskset id."""
|
|
79
|
-
serve: ServeConfig | None = Field(default_factory=ServeConfig)
|
|
80
|
-
"""How the env is hosted: the env-server worker pool (elastic by default) and each
|
|
81
|
-
worker's episode bound — the path prime-rl trains through. `--no-serve` runs the
|
|
82
|
-
rollouts in-process instead."""
|
|
83
78
|
run: RunConfig = Field(default_factory=RunConfig)
|
|
84
79
|
"""Run identity: `run.name` is the display name, `run.dir` names the directory
|
|
85
80
|
under `output_dir`, and `run.id` is stamped on traces."""
|
|
@@ -110,8 +105,7 @@ class EvalConfig(BaseConfig):
|
|
|
110
105
|
)
|
|
111
106
|
"""Episodes in flight at once, `None` for no limit. An episode plays its agents one
|
|
112
107
|
at a time, so this is the live agent runs too — until `--env.max-concurrent-agents`
|
|
113
|
-
says otherwise.
|
|
114
|
-
`--serve.max-concurrent` pins one."""
|
|
108
|
+
says otherwise."""
|
|
115
109
|
verbose: bool = Field(False, validation_alias=AliasChoices("verbose", "v"))
|
|
116
110
|
"""Log at debug level instead of the default info."""
|
|
117
111
|
dry_run: bool = Field(False, exclude=True)
|
|
@@ -122,8 +116,7 @@ class EvalConfig(BaseConfig):
|
|
|
122
116
|
previous run's results. Excluded from the saved config."""
|
|
123
117
|
rich: RichConfig | None = Field(default_factory=RichConfig)
|
|
124
118
|
"""The live dashboard (on by default; `--no-rich` streams logs to the console
|
|
125
|
-
instead).
|
|
126
|
-
each episode completes; `--rich.show-logs` swaps the rows for the run's logs."""
|
|
119
|
+
instead); `--rich.show-logs` swaps the rollout rows for the run's logs."""
|
|
127
120
|
push: bool = True
|
|
128
121
|
"""Upload the finished run to the Prime Intellect platform (the private Evaluations
|
|
129
122
|
tab) at the end of the eval. On by default; disable with `--no-push`. Needs
|
|
@@ -136,7 +129,7 @@ class EvalConfig(BaseConfig):
|
|
|
136
129
|
resume: bool = Field(False, exclude=True)
|
|
137
130
|
"""Re-run the run's missing/errored rollouts in place instead of starting fresh. The
|
|
138
131
|
run dir comes from the resolved config (`output_dir / run.dir`), so resume with the
|
|
139
|
-
run's own config — e.g. `uv run eval @ <run-dir>/configs/eval.json --resume`. Excluded
|
|
132
|
+
run's own config — e.g. `uv run vf-eval @ <run-dir>/configs/eval.json --resume`. Excluded
|
|
140
133
|
from the saved config."""
|
|
141
134
|
|
|
142
135
|
@model_validator(mode="before")
|
|
@@ -56,7 +56,7 @@ class ValidateConfig(BaseConfig):
|
|
|
56
56
|
resume: bool = Field(False, exclude=True)
|
|
57
57
|
"""Re-run the run's missing, errored, and timed-out tasks in place. The run dir comes
|
|
58
58
|
from the resolved config (`output_dir / run.dir`), so resume with the run's own
|
|
59
|
-
config — e.g. `uv run validate @ <run-dir>/configs/validate.json --resume`.
|
|
59
|
+
config — e.g. `uv run vf-validate @ <run-dir>/configs/validate.json --resume`.
|
|
60
60
|
Excluded from the saved config."""
|
|
61
61
|
clean: bool = Field(False, exclude=True)
|
|
62
62
|
"""Delete the run directory (`output_dir / run.dir`) before running, overwriting a
|
verifiers/v1/configs/client.py
CHANGED
|
@@ -4,7 +4,7 @@ A `BaseClientConfig` is an OpenAI-compatible endpoint (base_url + API-key env va
|
|
|
4
4
|
+ extra headers); `clients.resolve_client` turns one into a live `Client` — the
|
|
5
5
|
interception server builds one per distinct config and shares it across the rollouts
|
|
6
6
|
it multiplexes. The default Prime endpoint, API key, and team fall back to
|
|
7
|
-
the active Prime CLI config, so direct `uv run eval` calls behave like `prime eval`.
|
|
7
|
+
the active Prime CLI config, so direct `uv run vf-eval` calls behave like `prime eval`.
|
|
8
8
|
Both the eval entrypoint (its model client) and in-env LLM calls (e.g. a judge reward)
|
|
9
9
|
build clients from these. `ClientConfig` is the CLI-selectable discriminated union
|
|
10
10
|
(eval | train).
|
verifiers/v1/env.py
CHANGED
|
@@ -316,17 +316,24 @@ class Env(ABC, Generic[ConfigT]):
|
|
|
316
316
|
ctx: ModelContext,
|
|
317
317
|
semaphore: asyncio.Semaphore | None = None,
|
|
318
318
|
on_complete: Callable[[Episode], Awaitable[None]] | None = None,
|
|
319
|
+
on_trace: Callable[[Trace], None] | None = None,
|
|
319
320
|
) -> Episode:
|
|
320
321
|
"""Run one planned episode to completion, with whole-episode
|
|
321
322
|
retries per `--env.retries`; `semaphore` bounds concurrent EPISODES — one
|
|
322
323
|
permit for the attempt in flight, held across the whole of it (its agents,
|
|
323
324
|
their boxes, `finalize()`) and released before a retry's backoff and before
|
|
324
|
-
`on_complete` (the runners' persistence hook, which fires when final).
|
|
325
|
+
`on_complete` (the runners' persistence hook, which fires when final).
|
|
326
|
+
`on_trace` sees each trace at mint, after it joined `slot.traces`."""
|
|
325
327
|
|
|
326
328
|
async def attempt() -> Episode:
|
|
327
329
|
slot.traces = [] # a retry shows the fresh attempt's traces
|
|
328
330
|
live = slot.traces
|
|
329
331
|
|
|
332
|
+
def minted(trace: Trace) -> None:
|
|
333
|
+
live.append(trace)
|
|
334
|
+
if on_trace is not None:
|
|
335
|
+
on_trace(trace)
|
|
336
|
+
|
|
330
337
|
def discard(trace: Trace) -> None:
|
|
331
338
|
# A retried agent attempt abandons its trace; drop it from the view.
|
|
332
339
|
with contextlib.suppress(ValueError):
|
|
@@ -336,7 +343,7 @@ class Env(ABC, Generic[ConfigT]):
|
|
|
336
343
|
return await self.run_episode(
|
|
337
344
|
slot.task,
|
|
338
345
|
ctx,
|
|
339
|
-
on_trace=
|
|
346
|
+
on_trace=minted,
|
|
340
347
|
on_discard=discard,
|
|
341
348
|
)
|
|
342
349
|
|
verifiers/v1/graph.py
CHANGED
|
@@ -548,8 +548,13 @@ class PendingTurn:
|
|
|
548
548
|
"""Add this turn to the graph; returns the committed assistant node's id."""
|
|
549
549
|
assistant_id = _commit_turn(self, response)
|
|
550
550
|
self.trace.tools = self.tools
|
|
551
|
+
self.trace.clear_preview(self)
|
|
551
552
|
return assistant_id
|
|
552
553
|
|
|
554
|
+
def abandon(self) -> None:
|
|
555
|
+
"""The request failed or was cancelled before a commit: its preview goes."""
|
|
556
|
+
self.trace.clear_preview(self)
|
|
557
|
+
|
|
553
558
|
def commit_prompt(self) -> None:
|
|
554
559
|
"""Record an input that terminated before model inference."""
|
|
555
560
|
parent = self.prefix_node_ids[-1] if self.prefix_node_ids else None
|
|
@@ -570,6 +575,7 @@ class PendingTurn:
|
|
|
570
575
|
parent = len(self.trace.nodes) - 1
|
|
571
576
|
index[_node_key(previous, message, self.tools)] = parent
|
|
572
577
|
self.trace.tools = self.tools
|
|
578
|
+
self.trace.clear_preview(self)
|
|
573
579
|
|
|
574
580
|
|
|
575
581
|
def prepare_turn(
|
|
@@ -68,16 +68,16 @@ class CompactionConfig(BaseConfig):
|
|
|
68
68
|
|
|
69
69
|
class RLMHarnessConfig(HarnessConfig):
|
|
70
70
|
version: str = Field(
|
|
71
|
-
default="
|
|
71
|
+
default="00d8aa713c820023033bce6aa2eacea6c37e675d", min_length=1
|
|
72
72
|
)
|
|
73
73
|
"""Git ref (branch, tag, or commit) of nano-rlm to install. Must know every
|
|
74
74
|
field this harness puts on the wire, i.e. be at least the default ref."""
|
|
75
75
|
max_depth: NonNegativeInt | None = None
|
|
76
76
|
"""Recursion depth RLM may spawn sub-agents to; `None` = nano-rlm's default (1).
|
|
77
77
|
Set 0 to disable recursion."""
|
|
78
|
-
builtin_skills: list[BuiltinSkill] = Field(default_factory=
|
|
79
|
-
"""Built-in rlm skills to enable (the contract's `skills`)
|
|
80
|
-
|
|
78
|
+
builtin_skills: list[BuiltinSkill] = Field(default_factory=lambda: ["edit"])
|
|
79
|
+
"""Built-in rlm skills to enable (the contract's `skills`); `["edit"]` by default,
|
|
80
|
+
`[]` enables none. The base `skills` field takes SKILL.md paths."""
|
|
81
81
|
builtin_tools: list[BuiltinTool] | None = None
|
|
82
82
|
"""Native tool selection; None uses nano-rlm's default and is omitted from the
|
|
83
83
|
runtime contract. Explicit selection requires a nano-rlm ref supporting it."""
|
|
@@ -90,7 +90,7 @@ class RLMHarnessConfig(HarnessConfig):
|
|
|
90
90
|
"""Sub-agents running at once per session tree; `None` = nano-rlm's default (4),
|
|
91
91
|
raised to an explicit `max_depth` when needed to keep the policy valid."""
|
|
92
92
|
max_subagent_calls: PositiveInt | None = None
|
|
93
|
-
"""Tree-total recursive call cap; `None` uses nano-rlm's default (
|
|
93
|
+
"""Tree-total recursive call cap; `None` uses nano-rlm's default (uncapped)."""
|
|
94
94
|
exec_timeout: PositiveInt | None = None
|
|
95
95
|
"""IPython/native tool execution timeout in seconds; `None` uses nano-rlm's
|
|
96
96
|
default (300). Separate from `tool_timeout`, which controls MCP calls."""
|
|
@@ -101,10 +101,11 @@ class RLMHarnessConfig(HarnessConfig):
|
|
|
101
101
|
max_total_turns: PositiveInt | None = None
|
|
102
102
|
"""Tree-total turn budget (one turn = one work-loop model call, any engine); every
|
|
103
103
|
engine stops before its next call once spent. `None` = uncapped."""
|
|
104
|
-
max_total_tokens:
|
|
104
|
+
max_total_tokens: NonNegativeInt | None = 10_000_000
|
|
105
105
|
"""Tree-total budget of NEW tokens (completion + uncached prompt) across the session
|
|
106
|
-
tree; once spent every engine stops and no further sub-agents spawn.
|
|
107
|
-
|
|
106
|
+
tree; once spent every engine stops and no further sub-agents spawn. 10M by default;
|
|
107
|
+
`0` removes the budget (the rollout timeout is then the only terminator); `None`
|
|
108
|
+
defers to nano-rlm's own default (1,000,000)."""
|
|
108
109
|
max_tool_output_bytes: PositiveInt | None = None
|
|
109
110
|
"""Byte budget for a single tool result entering the conversation (middle truncation);
|
|
110
111
|
overrides rlm's built-in 20KB default in either direction."""
|
|
@@ -192,7 +193,7 @@ class RLMHarness(ACPHarness[RLMHarnessConfig]):
|
|
|
192
193
|
"exec_timeout": self.config.exec_timeout,
|
|
193
194
|
"allow_git": self.config.allow_git,
|
|
194
195
|
"max_total_turns": self.config.max_total_turns,
|
|
195
|
-
"max_total_tokens": self.config.max_total_tokens,
|
|
196
|
+
"max_total_tokens": self.config.max_total_tokens or None,
|
|
196
197
|
"max_tool_output_bytes": self.config.max_tool_output_bytes,
|
|
197
198
|
}
|
|
198
199
|
if isinstance(compaction, bool):
|
|
@@ -217,7 +218,13 @@ class RLMHarness(ACPHarness[RLMHarnessConfig]):
|
|
|
217
218
|
# None = passthrough: the key stays off the wire and nano-rlm's own
|
|
218
219
|
# default applies.
|
|
219
220
|
"policy": {
|
|
220
|
-
|
|
221
|
+
**{k: v for k, v in policy_knobs.items() if v is not None},
|
|
222
|
+
# explicit null: nano-rlm reads it as unbounded (its own default is 1M)
|
|
223
|
+
**(
|
|
224
|
+
{"max_total_tokens": None}
|
|
225
|
+
if self.config.max_total_tokens == 0
|
|
226
|
+
else {}
|
|
227
|
+
),
|
|
221
228
|
},
|
|
222
229
|
"system_prompt_path": None,
|
|
223
230
|
"append_to_system_prompt": "\n\n".join(appends) or None,
|