verifiers 0.3.2.dev99__py3-none-any.whl → 0.3.2.dev101__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
verifiers/v1/agent.py CHANGED
@@ -54,11 +54,19 @@ __all__ = ["Agent", "AgentConfig", "Agents", "TimeoutConfig", "make_agent"]
54
54
  logger = logging.getLogger(__name__)
55
55
 
56
56
 
57
+ DEFAULT_ROLLOUT_TIMEOUT = 4 * 3600.0
58
+ """Agent solve-attempt budget when neither the eval config nor the task sets one."""
59
+
60
+
57
61
  def resolve_rollout_timeouts(timeout: TimeoutConfig, task: Task) -> RolloutTimeouts:
58
62
  """Apply an agent's stage-timeout precedence to one task."""
59
63
  agent_timeout = (
60
64
  timeout.rollout if timeout.rollout is not None else task.data.timeout.agent
61
65
  )
66
+ if agent_timeout is None:
67
+ agent_timeout = DEFAULT_ROLLOUT_TIMEOUT
68
+ elif agent_timeout == 0:
69
+ agent_timeout = None # explicit: unbounded
62
70
  return RolloutTimeouts(
63
71
  setup=timeout.setup if timeout.setup is not None else task.data.timeout.setup,
64
72
  agent=agent_timeout,
@@ -16,6 +16,7 @@ from rich.table import Table
16
16
  from rich.text import Text
17
17
 
18
18
  from verifiers.v1.cli.dashboard.base import live_view
19
+ from verifiers.v1.cli.eval.hint import PRIME_RL_HINT
19
20
  from verifiers.v1.cli.output import attempt_log_file, output_path
20
21
  from verifiers.v1.configs.cli.eval import EvalConfig
21
22
  from verifiers.v1.env import RunSlot
@@ -280,6 +281,10 @@ def Overview(config: EvalConfig) -> Table:
280
281
  return grid
281
282
 
282
283
 
284
+ def _prime_rl_footer() -> Group:
285
+ return Group(Text(""), Text(PRIME_RL_HINT, style="dim", overflow="fold"))
286
+
287
+
283
288
  def _push_footer(push: "PushState | None") -> Group | None:
284
289
  """The `--push` line under the rollouts: dim with the run's URL while it streams,
285
290
  white once pushed, red when it failed. `None` when `--push` is off or the run stayed
@@ -833,13 +838,15 @@ def _render(
833
838
  # The --push status line (and, on Ctrl-C, the cleanup notice) appear under the rollouts. Measure
834
839
  # the fixed top (header + progress + rule) and the footer so the rollout rows fill what's left;
835
840
  # page through them (timer / arrows) when they'd overflow (else rich truncates).
836
- footers = [f for f in (_push_footer(push), _interrupt_footer()) if f is not None]
837
- footer = Group(*footers) if footers else None
841
+ footers = [
842
+ f
843
+ for f in (_push_footer(push), _interrupt_footer(), _prime_rl_footer())
844
+ if f is not None
845
+ ]
846
+ footer = Group(*footers)
838
847
  progress = Progress(slots, start, completed)
839
848
  top = Group(header, progress, Rule(style="dim"))
840
- reserved = len(_CONSOLE.render_lines(top))
841
- if footer is not None:
842
- reserved += len(_CONSOLE.render_lines(footer))
849
+ reserved = len(_CONSOLE.render_lines(top)) + len(_CONSOLE.render_lines(footer))
843
850
  rows_per_page = max(1, _CONSOLE.size.height - reserved - 1)
844
851
  if tail is not None: # --show-logs: the run's log stream in place of rollout rows
845
852
  parts = [
@@ -847,9 +854,8 @@ def _render(
847
854
  progress,
848
855
  Rule(style="dim"),
849
856
  tail.view(rows_per_page),
857
+ footer,
850
858
  ]
851
- if footer is not None:
852
- parts.append(footer)
853
859
  return Group(*parts)
854
860
  page_groups, index, count = _paginate(_groups(slots), rows_per_page, pager, now)
855
861
  if count > 1:
@@ -859,9 +865,8 @@ def _render(
859
865
  progress,
860
866
  Rule(style="dim"),
861
867
  Rows(page_groups, now, runtime_type, completed),
868
+ footer,
862
869
  ]
863
- if footer is not None:
864
- parts.append(footer)
865
870
  return Group(*parts)
866
871
 
867
872
 
verifiers/v1/cli/debug.py CHANGED
@@ -36,7 +36,7 @@ from verifiers.v1.utils.logging import setup_logging
36
36
  logger = logging.getLogger(__name__)
37
37
 
38
38
  USAGE = (
39
- "usage: uv run debug [<taskset-id>] (--command <cmd> | --script-path <path>) "
39
+ "usage: uv run vf-debug [<taskset-id>] (--command <cmd> | --script-path <path>) "
40
40
  "[--runtime.type subprocess] [options] [@ file.toml]\n"
41
41
  " runs setup, then one command or uploaded host script, and saves traces"
42
42
  )
@@ -0,0 +1,5 @@
1
+ PRIME_RL_HINT = (
2
+ "vf-eval will be deprecated - move to prime-rl's `uv run eval`: it streams traces "
3
+ "live into a dashboard, evaluates several envs in one run and scales through env "
4
+ "servers. https://github.com/PrimeIntellect-ai/prime-rl"
5
+ )
@@ -30,8 +30,8 @@ from verifiers.v1.utils.logging import setup_logging
30
30
  logger = logging.getLogger(__name__)
31
31
 
32
32
  USAGE = (
33
- "usage: uv run eval [<taskset-id>] [--env.id <id>] [options] [@ file.toml]\n"
34
- " uv run eval @ <run-dir>/configs/resolved/eval.json --resume (re-run the run's missing/errored rollouts)"
33
+ "usage: uv run vf-eval [<taskset-id>] [--env.id <id>] [options] [@ file.toml]\n"
34
+ " uv run vf-eval @ <run-dir>/configs/resolved/eval.json --resume (re-run the run's missing/errored rollouts)"
35
35
  )
36
36
 
37
37
 
@@ -48,7 +48,7 @@ def main(argv: list[str] | None = None) -> None:
48
48
  return
49
49
  # An env-block flag skips the usage gate so the typed parse renders its
50
50
  # did-you-mean instead of a bare usage line.
51
- typed_axis = any(a.startswith(("--env.", "--serve.")) for a in argv)
51
+ typed_axis = any(a.startswith("--env.") for a in argv)
52
52
  if (
53
53
  not extract_id(argv, "env.taskset")
54
54
  and not references_config_file(argv)
@@ -108,7 +108,7 @@ def main(argv: list[str] | None = None) -> None:
108
108
  raise SystemExit(
109
109
  f"--resume requires the exact config the run was started with - it "
110
110
  f"differs in [{', '.join(changed)}]. Resumed rollouts would not be "
111
- f"comparable; re-run with `uv run eval @ {saved_path} --resume`, or "
111
+ f"comparable; re-run with `uv run vf-eval @ {saved_path} --resume`, or "
112
112
  "start a fresh run"
113
113
  )
114
114
  if config.dry_run: # resolved + validated; write it to the output dir and exit
@@ -116,8 +116,7 @@ def main(argv: list[str] | None = None) -> None:
116
116
  logger.info("wrote config to %s", write_config(config, run_path))
117
117
  return
118
118
  # Always tee this attempt's logs to `logs/attempt_<n>/eval.log` (`logs/latest`
119
- # points there) — in server mode (the default) the workers write there too, and
120
- # `--rich.show-logs` tails it live.
119
+ # points there); `--rich.show-logs` tails it live.
121
120
  log_file = str(create_attempt_log_dir(run_path) / "eval.log")
122
121
  level = "DEBUG" if config.verbose else "INFO"
123
122
  setup_logging(level, log_file=log_file, console=config.rich is None)
@@ -128,7 +127,6 @@ def main(argv: list[str] | None = None) -> None:
128
127
  install_interrupt()
129
128
 
130
129
  try:
131
- # Through the env-server worker pool by default; in-process with --no-serve.
132
130
  episodes = asyncio.run(run_eval(config))
133
131
  except KeyboardInterrupt:
134
132
  # Graceful cleanup has already run (each rollout's `finally`); partial results are on
@@ -1,12 +1,9 @@
1
1
  """The eval runner: fan episodes out with bounded concurrency.
2
2
 
3
- Rollouts run through the env-server worker pool by default (`[serve]` sizes it;
4
- elastic — one worker, scaling on demand), the same path prime-rl trains through.
5
- `--no-serve` runs them in-process instead. Both paths share this runner — task
6
- selection, resume, persistence, the dashboard — and differ only in how one slot
7
- becomes one episode: `env.run_slot` in-process, a `run` request to the pool
8
- otherwise. The dashboard watches the same `RunSlot`s either way; a served slot
9
- has no live traces, so its per-turn detail lands when the episode completes.
3
+ Rollouts run in-process: the env comes up once in this process and each slot becomes
4
+ one episode through `env.run_slot`. The dashboard watches the same `RunSlot`s the env
5
+ fills with live traces. Serving an env to many consumers is prime-rl's job (its `eval`
6
+ runs env servers); this CLI is the quick local path.
10
7
  """
11
8
 
12
9
  import asyncio
@@ -18,16 +15,15 @@ from typing import TypeVar, cast
18
15
 
19
16
  from verifiers.v1.cli.dashboard import dashboard
20
17
  from verifiers.v1.cli.eval import resume
18
+ from verifiers.v1.cli.eval.hint import PRIME_RL_HINT
21
19
  from verifiers.v1.cli.output import (
22
20
  append_episode,
23
- attempt_log_file,
24
21
  output_path,
25
22
  save_config,
26
23
  )
27
24
  from verifiers.v1.cli.resume import distribute
28
25
  from verifiers.v1.clients import ModelContext
29
26
  from verifiers.v1.configs.cli.eval import EvalConfig
30
- from verifiers.v1.configs.serve import ServeConfig
31
27
  from verifiers.v1.env import Env, RunSlot
32
28
  from verifiers.v1.episode import Episode, EvalRunInfo
33
29
  from verifiers.v1.utils.aio import run_shielded
@@ -84,93 +80,11 @@ async def _in_process(
84
80
  yield run
85
81
 
86
82
 
87
- @contextlib.asynccontextmanager
88
- async def _server(
89
- config: EvalConfig,
90
- serve: ServeConfig,
91
- semaphore: asyncio.Semaphore | None,
92
- on_complete: OnComplete,
93
- ) -> AsyncIterator[RunSlotFn]:
94
- """Run slots through a spawned env-server worker pool: each rollout is its own
95
- `run` request, dispatched least-busy across workers. The workers own the env
96
- (and its serving resources); this process owns the taskset and the results."""
97
- import multiprocessing as mp
98
- from functools import partial
99
-
100
- from verifiers.v1.configs.serve import pool_serve_kwargs
101
- from verifiers.v1.serve import EnvClient, env_config_data, serve_env
102
- from verifiers.v1.utils.logging import setup_logging
103
-
104
- # Spawned processes inherit no logging — hand them the main process's setup so
105
- # their rollout logs land in the output dir. They share its stderr, so console
106
- # output follows the main process's choice: off under the dashboard (worker log
107
- # lines would print over the Live view and shift it), on otherwise.
108
- level = "DEBUG" if config.verbose else "INFO"
109
- log_file = str(attempt_log_file(output_path(config)))
110
- console = config.rich is None
111
- mpctx = mp.get_context("spawn")
112
- address_queue: mp.Queue = mpctx.Queue()
113
- # Death pipe: serve_env self-terminates if this process dies abruptly — we keep
114
- # parent_conn, whose close (even on our SIGKILL) signals the child's watch.
115
- parent_conn, child_conn = mpctx.Pipe()
116
- proc = mpctx.Process(
117
- target=serve_env,
118
- kwargs=dict(
119
- **pool_serve_kwargs(serve.pool),
120
- address="tcp://127.0.0.1:0",
121
- address_queue=address_queue,
122
- death_pipe=child_conn,
123
- log_setup=partial(setup_logging, level, log_file, console),
124
- config_data=env_config_data(config.env), # picklable across the spawn
125
- # `-c` seeds each worker's episode bound unless `[serve]` pins one — so a
126
- # pool carries `workers * bound` episodes, as `multiplex` implies.
127
- max_concurrent=serve.max_concurrent
128
- if serve.max_concurrent is not None
129
- else config.max_concurrent,
130
- ),
131
- daemon=False,
132
- )
133
- proc.start()
134
- child_conn.close() # the child holds its end; we keep parent_conn so our exit closes it
135
- try:
136
- address = await asyncio.to_thread(address_queue.get, timeout=600)
137
- client = EnvClient(address=address)
138
- try:
139
- await client.wait_for_server_startup(timeout=600)
140
-
141
- async def run(slot: RunSlot) -> Episode:
142
- async with semaphore or contextlib.nullcontext():
143
- slot.started = time.time()
144
- episode = await client.run(
145
- client=config.client,
146
- model=config.model,
147
- sampling=config.sampling,
148
- task_data=slot.task.data.model_dump(mode="json"),
149
- )
150
- slot.traces = list(episode.traces)
151
- slot.episode = cast(Episode, episode)
152
- slot.done = True
153
- await on_complete(cast(Episode, episode))
154
- return cast(Episode, episode)
155
-
156
- yield run
157
- finally:
158
- await client.close()
159
- finally:
160
- proc.terminate()
161
- with contextlib.suppress(Exception):
162
- await asyncio.to_thread(proc.join, 10)
163
- with contextlib.suppress(Exception):
164
- parent_conn.close()
165
-
166
-
167
83
  async def run_eval(config: EvalConfig) -> list[Episode]:
168
- from verifiers.v1.utils.loaders import load_environment, load_taskset
84
+ from verifiers.v1.utils.loaders import load_environment
169
85
 
170
- # The env comes up in this process only for an in-process run; a served run's
171
- # workers each load their own, and this process owns just the taskset.
172
- env = None if config.serve is not None else load_environment(config.env)
173
- taskset = env.taskset if env is not None else load_taskset(config.env.taskset)
86
+ env = load_environment(config.env)
87
+ taskset = env.taskset
174
88
  if config.num_tasks is None and taskset.INFINITE:
175
89
  raise ValueError(
176
90
  f"{type(taskset).__name__} is infinite - bound the run with -n"
@@ -186,14 +100,13 @@ async def run_eval(config: EvalConfig) -> list[Episode]:
186
100
  finished: list[Episode] = []
187
101
  if config.resume:
188
102
  keys = [task.hash for task in tasks]
189
- # In-process, the env's own keep-verdict decides what resumes; a served run
190
- # can't ask the worker-side env, so it keeps the default `episode.ok`.
191
- complete = (
192
- (lambda episode: env.complete(cast(Episode, episode)))
193
- if env is not None
194
- else None
103
+ # the env's own keep-verdict decides what resumes
104
+ loaded, owed = resume.load(
105
+ out,
106
+ keys,
107
+ config.num_rollouts,
108
+ lambda episode: env.complete(cast(Episode, episode)),
195
109
  )
196
- loaded, owed = resume.load(out, keys, config.num_rollouts, complete)
197
110
  finished = [cast(Episode, episode) for episode in loaded]
198
111
  if not owed: # already complete - report it and exit successfully
199
112
  print(
@@ -211,18 +124,10 @@ async def run_eval(config: EvalConfig) -> list[Episode]:
211
124
  )
212
125
  else:
213
126
  save_config(config, out)
214
- via = (
215
- f" via the env-server {config.serve.pool.type} pool"
216
- if config.serve is not None
217
- else ""
218
- )
219
127
  logger.info(
220
- "running %dx%d rollouts on %s%s",
221
- len(plan),
222
- config.num_rollouts,
223
- config.model,
224
- via,
128
+ "running %dx%d rollouts on %s", len(plan), config.num_rollouts, config.model
225
129
  )
130
+ logger.info(PRIME_RL_HINT)
226
131
  start = time.time()
227
132
  logger.info("results: %s", out)
228
133
 
@@ -242,25 +147,11 @@ async def run_eval(config: EvalConfig) -> list[Episode]:
242
147
  await append_episode(out, episode, write_lock)
243
148
  await asyncio.to_thread(log_episodes, run, [episode])
244
149
 
245
- backend = (
246
- _in_process(env, config, semaphore, on_complete)
247
- if env is not None
248
- else _server(config, config.serve, semaphore, on_complete)
249
- )
250
- # The run is closed out whatever breaks, backend setup and teardown included.
150
+ # The run is closed out whatever breaks, env setup and teardown included.
251
151
  try:
252
- async with backend as run_slot:
253
- # The display slots: in-process ones are the env's own (it fills their live
254
- # traces); a served rollout's is a client-side stand-in its worker never sees.
255
- planned = [
256
- slot
257
- for task, n in plan
258
- for slot in (
259
- env.slots(task, n)
260
- if env is not None
261
- else [RunSlot(task) for _ in range(n)]
262
- )
263
- ]
152
+ async with _in_process(env, config, semaphore, on_complete) as run_slot:
153
+ # the env's own slots: it fills their live traces as the rollouts run
154
+ planned = [slot for task, n in plan for slot in env.slots(task, n)]
264
155
  slots = [RunSlot.finished(episode) for episode in finished] + planned
265
156
  display = (
266
157
  dashboard(slots, config, start, push=push_state)
verifiers/v1/cli/gepa.py CHANGED
@@ -1,4 +1,4 @@
1
- """The GEPA entrypoint: `uv run gepa [<taskset-id>] --model <model> [options]`.
1
+ """The GEPA entrypoint: `uv run vf-gepa [<taskset-id>] --model <model> [options]`.
2
2
 
3
3
  Registered as the `gepa` console script. Optimizes a v1 taskset's `Task.system_prompt` via
4
4
  GEPA (Genetic-Pareto): alternating rollouts with a teacher LM reflecting on results — see
@@ -28,7 +28,7 @@ from verifiers.v1.utils.logging import setup_logging
28
28
 
29
29
  logger = logging.getLogger(__name__)
30
30
 
31
- USAGE = "usage: uv run gepa [<taskset-id>] [--env.id <id>] --model <model> [options] [@ file.toml]"
31
+ USAGE = "usage: uv run vf-gepa [<taskset-id>] [--env.id <id>] --model <model> [options] [@ file.toml]"
32
32
 
33
33
 
34
34
  def main(argv: list[str] | None = None) -> None:
verifiers/v1/cli/init.py CHANGED
@@ -8,7 +8,7 @@ from pydantic_config import cli
8
8
  from verifiers.v1.configs.cli.init import InitConfig
9
9
 
10
10
  USAGE = (
11
- "usage: uv run init <name> [--path ./environments] [-T/--add-tool] "
11
+ "usage: uv run vf-init <name> [--path ./environments] [-T/--add-tool] "
12
12
  "[-H/--add-harness]\n"
13
13
  " scaffold a new environment package"
14
14
  )
@@ -191,7 +191,7 @@ A v1 verifiers environment, scaffolded with `init`.
191
191
 
192
192
  ```bash
193
193
  uv pip install -e . # install this package (or register it in your project)
194
- uv run eval {dash} -n 3 # evaluate a few tasks with the bash harness
194
+ uv run vf-eval {dash} -n 3 # evaluate a few tasks with the bash harness
195
195
  ```
196
196
 
197
197
  ## Layout
@@ -228,7 +228,7 @@ def scaffold(config: InitConfig) -> Path:
228
228
  _write(pkg_dir / "servers" / "__init__.py", "")
229
229
  _write(pkg_dir / "servers" / "tool.py", _tool_py(stem, prefix))
230
230
 
231
- print(f"\ndone. next:\n uv pip install -e {env_dir}\n uv run eval {dash} -n 3")
231
+ print(f"\ndone. next:\n uv pip install -e {env_dir}\n uv run vf-eval {dash} -n 3")
232
232
  return env_dir
233
233
 
234
234
 
@@ -39,7 +39,7 @@ from verifiers.v1.utils.logging import setup_logging
39
39
  logger = logging.getLogger(__name__)
40
40
 
41
41
  USAGE = (
42
- "usage: uv run replay <output-dir> [options] [@ file.toml]\n"
42
+ "usage: uv run vf-replay <output-dir> [options] [@ file.toml]\n"
43
43
  " re-score a finished run's saved traces (judges + trace-only signals; no runtime)"
44
44
  )
45
45
 
@@ -49,9 +49,9 @@ REASONS = ("valid", "invalid", "unchecked", "error", "timeout")
49
49
  ResultRow = dict[str, Any]
50
50
 
51
51
  USAGE = (
52
- "usage: uv run validate [<taskset-id>] [--only-setup | --only-gold] "
52
+ "usage: uv run vf-validate [<taskset-id>] [--only-setup | --only-gold] "
53
53
  "[-o <output-dir>] [--runtime.type subprocess] [options] [@ file.toml]\n"
54
- " uv run validate @ <run-dir>/configs/validate.json --resume (re-run missing/errored/timed-out tasks)\n"
54
+ " uv run vf-validate @ <run-dir>/configs/validate.json --resume (re-run missing/errored/timed-out tasks)\n"
55
55
  " runs persisted gold and setup-only checks per task (no model)"
56
56
  )
57
57
 
@@ -16,7 +16,8 @@ class TimeoutConfig(BaseConfig):
16
16
  setup: float | None = None # one shared budget: task setup + provisioning
17
17
  """Timeout (in seconds) for the task + harness setup hooks."""
18
18
  rollout: float | None = None
19
- """Timeout (in seconds) for the agent's solve attempt."""
19
+ """Timeout (in seconds) for the agent's solve attempt. Unset: the task's own
20
+ timeout, else 4 hours. `0` disables the timeout."""
20
21
  finalize: float | None = None
21
22
  """Timeout (in seconds) for the task + harness finalize hooks."""
22
23
  scoring: float | None = None
@@ -9,7 +9,6 @@ from pydantic_config import BaseConfig
9
9
  from verifiers.v1.clients import ClientConfig, EvalClientConfig
10
10
  from verifiers.v1.configs.cli.env import narrowed_env_annotation, resolve_env_field
11
11
  from verifiers.v1.configs.env import EnvConfig
12
- from verifiers.v1.configs.serve import ServeConfig
13
12
  from verifiers.v1.envs.single_agent import SingleAgentEnvConfig
14
13
  from verifiers.v1.types import SamplingConfig
15
14
 
@@ -76,10 +75,6 @@ class EvalConfig(BaseConfig):
76
75
  env: SerializeAsAny[EnvConfig] = SingleAgentEnvConfig()
77
76
  """The environment — which env, its seed taskset, each agent, its knobs. Narrowed to
78
77
  the selected env's config class by the env id, else the taskset id."""
79
- serve: ServeConfig | None = Field(default_factory=ServeConfig)
80
- """How the env is hosted: the env-server worker pool (elastic by default) and each
81
- worker's episode bound — the path prime-rl trains through. `--no-serve` runs the
82
- rollouts in-process instead."""
83
78
  run: RunConfig = Field(default_factory=RunConfig)
84
79
  """Run identity: `run.name` is the display name, `run.dir` names the directory
85
80
  under `output_dir`, and `run.id` is stamped on traces."""
@@ -110,8 +105,7 @@ class EvalConfig(BaseConfig):
110
105
  )
111
106
  """Episodes in flight at once, `None` for no limit. An episode plays its agents one
112
107
  at a time, so this is the live agent runs too — until `--env.max-concurrent-agents`
113
- says otherwise. Under `[serve]` it also seeds each worker's bound, unless
114
- `--serve.max-concurrent` pins one."""
108
+ says otherwise."""
115
109
  verbose: bool = Field(False, validation_alias=AliasChoices("verbose", "v"))
116
110
  """Log at debug level instead of the default info."""
117
111
  dry_run: bool = Field(False, exclude=True)
@@ -122,8 +116,7 @@ class EvalConfig(BaseConfig):
122
116
  previous run's results. Excluded from the saved config."""
123
117
  rich: RichConfig | None = Field(default_factory=RichConfig)
124
118
  """The live dashboard (on by default; `--no-rich` streams logs to the console
125
- instead). A served run has no live per-turn view, so its rollout rows fill in as
126
- each episode completes; `--rich.show-logs` swaps the rows for the run's logs."""
119
+ instead); `--rich.show-logs` swaps the rollout rows for the run's logs."""
127
120
  push: bool = True
128
121
  """Upload the finished run to the Prime Intellect platform (the private Evaluations
129
122
  tab) at the end of the eval. On by default; disable with `--no-push`. Needs
@@ -136,7 +129,7 @@ class EvalConfig(BaseConfig):
136
129
  resume: bool = Field(False, exclude=True)
137
130
  """Re-run the run's missing/errored rollouts in place instead of starting fresh. The
138
131
  run dir comes from the resolved config (`output_dir / run.dir`), so resume with the
139
- run's own config — e.g. `uv run eval @ <run-dir>/configs/eval.json --resume`. Excluded
132
+ run's own config — e.g. `uv run vf-eval @ <run-dir>/configs/eval.json --resume`. Excluded
140
133
  from the saved config."""
141
134
 
142
135
  @model_validator(mode="before")
@@ -56,7 +56,7 @@ class ValidateConfig(BaseConfig):
56
56
  resume: bool = Field(False, exclude=True)
57
57
  """Re-run the run's missing, errored, and timed-out tasks in place. The run dir comes
58
58
  from the resolved config (`output_dir / run.dir`), so resume with the run's own
59
- config — e.g. `uv run validate @ <run-dir>/configs/validate.json --resume`.
59
+ config — e.g. `uv run vf-validate @ <run-dir>/configs/validate.json --resume`.
60
60
  Excluded from the saved config."""
61
61
  clean: bool = Field(False, exclude=True)
62
62
  """Delete the run directory (`output_dir / run.dir`) before running, overwriting a
@@ -4,7 +4,7 @@ A `BaseClientConfig` is an OpenAI-compatible endpoint (base_url + API-key env va
4
4
  + extra headers); `clients.resolve_client` turns one into a live `Client` — the
5
5
  interception server builds one per distinct config and shares it across the rollouts
6
6
  it multiplexes. The default Prime endpoint, API key, and team fall back to
7
- the active Prime CLI config, so direct `uv run eval` calls behave like `prime eval`.
7
+ the active Prime CLI config, so direct `uv run vf-eval` calls behave like `prime eval`.
8
8
  Both the eval entrypoint (its model client) and in-env LLM calls (e.g. a judge reward)
9
9
  build clients from these. `ClientConfig` is the CLI-selectable discriminated union
10
10
  (eval | train).
verifiers/v1/env.py CHANGED
@@ -316,17 +316,24 @@ class Env(ABC, Generic[ConfigT]):
316
316
  ctx: ModelContext,
317
317
  semaphore: asyncio.Semaphore | None = None,
318
318
  on_complete: Callable[[Episode], Awaitable[None]] | None = None,
319
+ on_trace: Callable[[Trace], None] | None = None,
319
320
  ) -> Episode:
320
321
  """Run one planned episode to completion, with whole-episode
321
322
  retries per `--env.retries`; `semaphore` bounds concurrent EPISODES — one
322
323
  permit for the attempt in flight, held across the whole of it (its agents,
323
324
  their boxes, `finalize()`) and released before a retry's backoff and before
324
- `on_complete` (the runners' persistence hook, which fires when final)."""
325
+ `on_complete` (the runners' persistence hook, which fires when final).
326
+ `on_trace` sees each trace at mint, after it joined `slot.traces`."""
325
327
 
326
328
  async def attempt() -> Episode:
327
329
  slot.traces = [] # a retry shows the fresh attempt's traces
328
330
  live = slot.traces
329
331
 
332
+ def minted(trace: Trace) -> None:
333
+ live.append(trace)
334
+ if on_trace is not None:
335
+ on_trace(trace)
336
+
330
337
  def discard(trace: Trace) -> None:
331
338
  # A retried agent attempt abandons its trace; drop it from the view.
332
339
  with contextlib.suppress(ValueError):
@@ -336,7 +343,7 @@ class Env(ABC, Generic[ConfigT]):
336
343
  return await self.run_episode(
337
344
  slot.task,
338
345
  ctx,
339
- on_trace=live.append,
346
+ on_trace=minted,
340
347
  on_discard=discard,
341
348
  )
342
349
 
verifiers/v1/graph.py CHANGED
@@ -548,8 +548,13 @@ class PendingTurn:
548
548
  """Add this turn to the graph; returns the committed assistant node's id."""
549
549
  assistant_id = _commit_turn(self, response)
550
550
  self.trace.tools = self.tools
551
+ self.trace.clear_preview(self)
551
552
  return assistant_id
552
553
 
554
+ def abandon(self) -> None:
555
+ """The request failed or was cancelled before a commit: its preview goes."""
556
+ self.trace.clear_preview(self)
557
+
553
558
  def commit_prompt(self) -> None:
554
559
  """Record an input that terminated before model inference."""
555
560
  parent = self.prefix_node_ids[-1] if self.prefix_node_ids else None
@@ -570,6 +575,7 @@ class PendingTurn:
570
575
  parent = len(self.trace.nodes) - 1
571
576
  index[_node_key(previous, message, self.tools)] = parent
572
577
  self.trace.tools = self.tools
578
+ self.trace.clear_preview(self)
573
579
 
574
580
 
575
581
  def prepare_turn(
@@ -68,16 +68,16 @@ class CompactionConfig(BaseConfig):
68
68
 
69
69
  class RLMHarnessConfig(HarnessConfig):
70
70
  version: str = Field(
71
- default="ad081dbcf5e8c1d4e5b431b4b7d4dd5f30b7367c", min_length=1
71
+ default="00d8aa713c820023033bce6aa2eacea6c37e675d", min_length=1
72
72
  )
73
73
  """Git ref (branch, tag, or commit) of nano-rlm to install. Must know every
74
74
  field this harness puts on the wire, i.e. be at least the default ref."""
75
75
  max_depth: NonNegativeInt | None = None
76
76
  """Recursion depth RLM may spawn sub-agents to; `None` = nano-rlm's default (1).
77
77
  Set 0 to disable recursion."""
78
- builtin_skills: list[BuiltinSkill] = Field(default_factory=list)
79
- """Built-in rlm skills to enable (the contract's `skills`), e.g. `["edit"]`;
80
- empty enables none. The base `skills` field takes SKILL.md paths."""
78
+ builtin_skills: list[BuiltinSkill] = Field(default_factory=lambda: ["edit"])
79
+ """Built-in rlm skills to enable (the contract's `skills`); `["edit"]` by default,
80
+ `[]` enables none. The base `skills` field takes SKILL.md paths."""
81
81
  builtin_tools: list[BuiltinTool] | None = None
82
82
  """Native tool selection; None uses nano-rlm's default and is omitted from the
83
83
  runtime contract. Explicit selection requires a nano-rlm ref supporting it."""
@@ -90,7 +90,7 @@ class RLMHarnessConfig(HarnessConfig):
90
90
  """Sub-agents running at once per session tree; `None` = nano-rlm's default (4),
91
91
  raised to an explicit `max_depth` when needed to keep the policy valid."""
92
92
  max_subagent_calls: PositiveInt | None = None
93
- """Tree-total recursive call cap; `None` uses nano-rlm's default (64)."""
93
+ """Tree-total recursive call cap; `None` uses nano-rlm's default (uncapped)."""
94
94
  exec_timeout: PositiveInt | None = None
95
95
  """IPython/native tool execution timeout in seconds; `None` uses nano-rlm's
96
96
  default (300). Separate from `tool_timeout`, which controls MCP calls."""
@@ -101,10 +101,11 @@ class RLMHarnessConfig(HarnessConfig):
101
101
  max_total_turns: PositiveInt | None = None
102
102
  """Tree-total turn budget (one turn = one work-loop model call, any engine); every
103
103
  engine stops before its next call once spent. `None` = uncapped."""
104
- max_total_tokens: PositiveInt | None = None
104
+ max_total_tokens: NonNegativeInt | None = 10_000_000
105
105
  """Tree-total budget of NEW tokens (completion + uncached prompt) across the session
106
- tree; once spent every engine stops and no further sub-agents spawn. `None` uses
107
- nano-rlm's default (1,000,000); it does not disable the budget."""
106
+ tree; once spent every engine stops and no further sub-agents spawn. 10M by default;
107
+ `0` removes the budget (the rollout timeout is then the only terminator); `None`
108
+ defers to nano-rlm's own default (1,000,000)."""
108
109
  max_tool_output_bytes: PositiveInt | None = None
109
110
  """Byte budget for a single tool result entering the conversation (middle truncation);
110
111
  overrides rlm's built-in 20KB default in either direction."""
@@ -192,7 +193,7 @@ class RLMHarness(ACPHarness[RLMHarnessConfig]):
192
193
  "exec_timeout": self.config.exec_timeout,
193
194
  "allow_git": self.config.allow_git,
194
195
  "max_total_turns": self.config.max_total_turns,
195
- "max_total_tokens": self.config.max_total_tokens,
196
+ "max_total_tokens": self.config.max_total_tokens or None,
196
197
  "max_tool_output_bytes": self.config.max_tool_output_bytes,
197
198
  }
198
199
  if isinstance(compaction, bool):
@@ -217,7 +218,13 @@ class RLMHarness(ACPHarness[RLMHarnessConfig]):
217
218
  # None = passthrough: the key stays off the wire and nano-rlm's own
218
219
  # default applies.
219
220
  "policy": {
220
- key: value for key, value in policy_knobs.items() if value is not None
221
+ **{k: v for k, v in policy_knobs.items() if v is not None},
222
+ # explicit null: nano-rlm reads it as unbounded (its own default is 1M)
223
+ **(
224
+ {"max_total_tokens": None}
225
+ if self.config.max_total_tokens == 0
226
+ else {}
227
+ ),
221
228
  },
222
229
  "system_prompt_path": None,
223
230
  "append_to_system_prompt": "\n\n".join(appends) or None,