verifiers 0.3.2.dev100__py3-none-any.whl → 0.3.2.dev101__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -16,6 +16,7 @@ from rich.table import Table
16
16
  from rich.text import Text
17
17
 
18
18
  from verifiers.v1.cli.dashboard.base import live_view
19
+ from verifiers.v1.cli.eval.hint import PRIME_RL_HINT
19
20
  from verifiers.v1.cli.output import attempt_log_file, output_path
20
21
  from verifiers.v1.configs.cli.eval import EvalConfig
21
22
  from verifiers.v1.env import RunSlot
@@ -280,6 +281,10 @@ def Overview(config: EvalConfig) -> Table:
280
281
  return grid
281
282
 
282
283
 
284
+ def _prime_rl_footer() -> Group:
285
+ return Group(Text(""), Text(PRIME_RL_HINT, style="dim", overflow="fold"))
286
+
287
+
283
288
  def _push_footer(push: "PushState | None") -> Group | None:
284
289
  """The `--push` line under the rollouts: dim with the run's URL while it streams,
285
290
  white once pushed, red when it failed. `None` when `--push` is off or the run stayed
@@ -833,13 +838,15 @@ def _render(
833
838
  # The --push status line (and, on Ctrl-C, the cleanup notice) appear under the rollouts. Measure
834
839
  # the fixed top (header + progress + rule) and the footer so the rollout rows fill what's left;
835
840
  # page through them (timer / arrows) when they'd overflow (else rich truncates).
836
- footers = [f for f in (_push_footer(push), _interrupt_footer()) if f is not None]
837
- footer = Group(*footers) if footers else None
841
+ footers = [
842
+ f
843
+ for f in (_push_footer(push), _interrupt_footer(), _prime_rl_footer())
844
+ if f is not None
845
+ ]
846
+ footer = Group(*footers)
838
847
  progress = Progress(slots, start, completed)
839
848
  top = Group(header, progress, Rule(style="dim"))
840
- reserved = len(_CONSOLE.render_lines(top))
841
- if footer is not None:
842
- reserved += len(_CONSOLE.render_lines(footer))
849
+ reserved = len(_CONSOLE.render_lines(top)) + len(_CONSOLE.render_lines(footer))
843
850
  rows_per_page = max(1, _CONSOLE.size.height - reserved - 1)
844
851
  if tail is not None: # --show-logs: the run's log stream in place of rollout rows
845
852
  parts = [
@@ -847,9 +854,8 @@ def _render(
847
854
  progress,
848
855
  Rule(style="dim"),
849
856
  tail.view(rows_per_page),
857
+ footer,
850
858
  ]
851
- if footer is not None:
852
- parts.append(footer)
853
859
  return Group(*parts)
854
860
  page_groups, index, count = _paginate(_groups(slots), rows_per_page, pager, now)
855
861
  if count > 1:
@@ -859,9 +865,8 @@ def _render(
859
865
  progress,
860
866
  Rule(style="dim"),
861
867
  Rows(page_groups, now, runtime_type, completed),
868
+ footer,
862
869
  ]
863
- if footer is not None:
864
- parts.append(footer)
865
870
  return Group(*parts)
866
871
 
867
872
 
verifiers/v1/cli/debug.py CHANGED
@@ -36,7 +36,7 @@ from verifiers.v1.utils.logging import setup_logging
36
36
  logger = logging.getLogger(__name__)
37
37
 
38
38
  USAGE = (
39
- "usage: uv run debug [<taskset-id>] (--command <cmd> | --script-path <path>) "
39
+ "usage: uv run vf-debug [<taskset-id>] (--command <cmd> | --script-path <path>) "
40
40
  "[--runtime.type subprocess] [options] [@ file.toml]\n"
41
41
  " runs setup, then one command or uploaded host script, and saves traces"
42
42
  )
@@ -0,0 +1,5 @@
1
+ PRIME_RL_HINT = (
2
+ "vf-eval will be deprecated - move to prime-rl's `uv run eval`: it streams traces "
3
+ "live into a dashboard, evaluates several envs in one run and scales through env "
4
+ "servers. https://github.com/PrimeIntellect-ai/prime-rl"
5
+ )
@@ -30,8 +30,8 @@ from verifiers.v1.utils.logging import setup_logging
30
30
  logger = logging.getLogger(__name__)
31
31
 
32
32
  USAGE = (
33
- "usage: uv run eval [<taskset-id>] [--env.id <id>] [options] [@ file.toml]\n"
34
- " uv run eval @ <run-dir>/configs/resolved/eval.json --resume (re-run the run's missing/errored rollouts)"
33
+ "usage: uv run vf-eval [<taskset-id>] [--env.id <id>] [options] [@ file.toml]\n"
34
+ " uv run vf-eval @ <run-dir>/configs/resolved/eval.json --resume (re-run the run's missing/errored rollouts)"
35
35
  )
36
36
 
37
37
 
@@ -48,7 +48,7 @@ def main(argv: list[str] | None = None) -> None:
48
48
  return
49
49
  # An env-block flag skips the usage gate so the typed parse renders its
50
50
  # did-you-mean instead of a bare usage line.
51
- typed_axis = any(a.startswith(("--env.", "--serve.")) for a in argv)
51
+ typed_axis = any(a.startswith("--env.") for a in argv)
52
52
  if (
53
53
  not extract_id(argv, "env.taskset")
54
54
  and not references_config_file(argv)
@@ -108,7 +108,7 @@ def main(argv: list[str] | None = None) -> None:
108
108
  raise SystemExit(
109
109
  f"--resume requires the exact config the run was started with - it "
110
110
  f"differs in [{', '.join(changed)}]. Resumed rollouts would not be "
111
- f"comparable; re-run with `uv run eval @ {saved_path} --resume`, or "
111
+ f"comparable; re-run with `uv run vf-eval @ {saved_path} --resume`, or "
112
112
  "start a fresh run"
113
113
  )
114
114
  if config.dry_run: # resolved + validated; write it to the output dir and exit
@@ -116,8 +116,7 @@ def main(argv: list[str] | None = None) -> None:
116
116
  logger.info("wrote config to %s", write_config(config, run_path))
117
117
  return
118
118
  # Always tee this attempt's logs to `logs/attempt_<n>/eval.log` (`logs/latest`
119
- # points there) — in server mode (the default) the workers write there too, and
120
- # `--rich.show-logs` tails it live.
119
+ # points there); `--rich.show-logs` tails it live.
121
120
  log_file = str(create_attempt_log_dir(run_path) / "eval.log")
122
121
  level = "DEBUG" if config.verbose else "INFO"
123
122
  setup_logging(level, log_file=log_file, console=config.rich is None)
@@ -128,7 +127,6 @@ def main(argv: list[str] | None = None) -> None:
128
127
  install_interrupt()
129
128
 
130
129
  try:
131
- # Through the env-server worker pool by default; in-process with --no-serve.
132
130
  episodes = asyncio.run(run_eval(config))
133
131
  except KeyboardInterrupt:
134
132
  # Graceful cleanup has already run (each rollout's `finally`); partial results are on
@@ -1,12 +1,9 @@
1
1
  """The eval runner: fan episodes out with bounded concurrency.
2
2
 
3
- Rollouts run through the env-server worker pool by default (`[serve]` sizes it;
4
- elastic — one worker, scaling on demand), the same path prime-rl trains through.
5
- `--no-serve` runs them in-process instead. Both paths share this runner — task
6
- selection, resume, persistence, the dashboard — and differ only in how one slot
7
- becomes one episode: `env.run_slot` in-process, a `run` request to the pool
8
- otherwise. The dashboard watches the same `RunSlot`s either way; a served slot
9
- has no live traces, so its per-turn detail lands when the episode completes.
3
+ Rollouts run in-process: the env comes up once in this process and each slot becomes
4
+ one episode through `env.run_slot`. The dashboard watches the same `RunSlot`s the env
5
+ fills with live traces. Serving an env to many consumers is prime-rl's job (its `eval`
6
+ runs env servers); this CLI is the quick local path.
10
7
  """
11
8
 
12
9
  import asyncio
@@ -18,16 +15,15 @@ from typing import TypeVar, cast
18
15
 
19
16
  from verifiers.v1.cli.dashboard import dashboard
20
17
  from verifiers.v1.cli.eval import resume
18
+ from verifiers.v1.cli.eval.hint import PRIME_RL_HINT
21
19
  from verifiers.v1.cli.output import (
22
20
  append_episode,
23
- attempt_log_file,
24
21
  output_path,
25
22
  save_config,
26
23
  )
27
24
  from verifiers.v1.cli.resume import distribute
28
25
  from verifiers.v1.clients import ModelContext
29
26
  from verifiers.v1.configs.cli.eval import EvalConfig
30
- from verifiers.v1.configs.serve import ServeConfig
31
27
  from verifiers.v1.env import Env, RunSlot
32
28
  from verifiers.v1.episode import Episode, EvalRunInfo
33
29
  from verifiers.v1.utils.aio import run_shielded
@@ -84,93 +80,11 @@ async def _in_process(
84
80
  yield run
85
81
 
86
82
 
87
- @contextlib.asynccontextmanager
88
- async def _server(
89
- config: EvalConfig,
90
- serve: ServeConfig,
91
- semaphore: asyncio.Semaphore | None,
92
- on_complete: OnComplete,
93
- ) -> AsyncIterator[RunSlotFn]:
94
- """Run slots through a spawned env-server worker pool: each rollout is its own
95
- `run` request, dispatched least-busy across workers. The workers own the env
96
- (and its serving resources); this process owns the taskset and the results."""
97
- import multiprocessing as mp
98
- from functools import partial
99
-
100
- from verifiers.v1.configs.serve import pool_serve_kwargs
101
- from verifiers.v1.serve import EnvClient, env_config_data, serve_env
102
- from verifiers.v1.utils.logging import setup_logging
103
-
104
- # Spawned processes inherit no logging — hand them the main process's setup so
105
- # their rollout logs land in the output dir. They share its stderr, so console
106
- # output follows the main process's choice: off under the dashboard (worker log
107
- # lines would print over the Live view and shift it), on otherwise.
108
- level = "DEBUG" if config.verbose else "INFO"
109
- log_file = str(attempt_log_file(output_path(config)))
110
- console = config.rich is None
111
- mpctx = mp.get_context("spawn")
112
- address_queue: mp.Queue = mpctx.Queue()
113
- # Death pipe: serve_env self-terminates if this process dies abruptly — we keep
114
- # parent_conn, whose close (even on our SIGKILL) signals the child's watch.
115
- parent_conn, child_conn = mpctx.Pipe()
116
- proc = mpctx.Process(
117
- target=serve_env,
118
- kwargs=dict(
119
- **pool_serve_kwargs(serve.pool),
120
- address="tcp://127.0.0.1:0",
121
- address_queue=address_queue,
122
- death_pipe=child_conn,
123
- log_setup=partial(setup_logging, level, log_file, console),
124
- config_data=env_config_data(config.env), # picklable across the spawn
125
- # `-c` seeds each worker's episode bound unless `[serve]` pins one — so a
126
- # pool carries `workers * bound` episodes, as `multiplex` implies.
127
- max_concurrent=serve.max_concurrent
128
- if serve.max_concurrent is not None
129
- else config.max_concurrent,
130
- ),
131
- daemon=False,
132
- )
133
- proc.start()
134
- child_conn.close() # the child holds its end; we keep parent_conn so our exit closes it
135
- try:
136
- address = await asyncio.to_thread(address_queue.get, timeout=600)
137
- client = EnvClient(address=address)
138
- try:
139
- await client.wait_for_server_startup(timeout=600)
140
-
141
- async def run(slot: RunSlot) -> Episode:
142
- async with semaphore or contextlib.nullcontext():
143
- slot.started = time.time()
144
- episode = await client.run(
145
- client=config.client,
146
- model=config.model,
147
- sampling=config.sampling,
148
- task_data=slot.task.data.model_dump(mode="json"),
149
- )
150
- slot.traces = list(episode.traces)
151
- slot.episode = cast(Episode, episode)
152
- slot.done = True
153
- await on_complete(cast(Episode, episode))
154
- return cast(Episode, episode)
155
-
156
- yield run
157
- finally:
158
- await client.close()
159
- finally:
160
- proc.terminate()
161
- with contextlib.suppress(Exception):
162
- await asyncio.to_thread(proc.join, 10)
163
- with contextlib.suppress(Exception):
164
- parent_conn.close()
165
-
166
-
167
83
  async def run_eval(config: EvalConfig) -> list[Episode]:
168
- from verifiers.v1.utils.loaders import load_environment, load_taskset
84
+ from verifiers.v1.utils.loaders import load_environment
169
85
 
170
- # The env comes up in this process only for an in-process run; a served run's
171
- # workers each load their own, and this process owns just the taskset.
172
- env = None if config.serve is not None else load_environment(config.env)
173
- taskset = env.taskset if env is not None else load_taskset(config.env.taskset)
86
+ env = load_environment(config.env)
87
+ taskset = env.taskset
174
88
  if config.num_tasks is None and taskset.INFINITE:
175
89
  raise ValueError(
176
90
  f"{type(taskset).__name__} is infinite - bound the run with -n"
@@ -186,14 +100,13 @@ async def run_eval(config: EvalConfig) -> list[Episode]:
186
100
  finished: list[Episode] = []
187
101
  if config.resume:
188
102
  keys = [task.hash for task in tasks]
189
- # In-process, the env's own keep-verdict decides what resumes; a served run
190
- # can't ask the worker-side env, so it keeps the default `episode.ok`.
191
- complete = (
192
- (lambda episode: env.complete(cast(Episode, episode)))
193
- if env is not None
194
- else None
103
+ # the env's own keep-verdict decides what resumes
104
+ loaded, owed = resume.load(
105
+ out,
106
+ keys,
107
+ config.num_rollouts,
108
+ lambda episode: env.complete(cast(Episode, episode)),
195
109
  )
196
- loaded, owed = resume.load(out, keys, config.num_rollouts, complete)
197
110
  finished = [cast(Episode, episode) for episode in loaded]
198
111
  if not owed: # already complete - report it and exit successfully
199
112
  print(
@@ -211,18 +124,10 @@ async def run_eval(config: EvalConfig) -> list[Episode]:
211
124
  )
212
125
  else:
213
126
  save_config(config, out)
214
- via = (
215
- f" via the env-server {config.serve.pool.type} pool"
216
- if config.serve is not None
217
- else ""
218
- )
219
127
  logger.info(
220
- "running %dx%d rollouts on %s%s",
221
- len(plan),
222
- config.num_rollouts,
223
- config.model,
224
- via,
128
+ "running %dx%d rollouts on %s", len(plan), config.num_rollouts, config.model
225
129
  )
130
+ logger.info(PRIME_RL_HINT)
226
131
  start = time.time()
227
132
  logger.info("results: %s", out)
228
133
 
@@ -242,25 +147,11 @@ async def run_eval(config: EvalConfig) -> list[Episode]:
242
147
  await append_episode(out, episode, write_lock)
243
148
  await asyncio.to_thread(log_episodes, run, [episode])
244
149
 
245
- backend = (
246
- _in_process(env, config, semaphore, on_complete)
247
- if env is not None
248
- else _server(config, config.serve, semaphore, on_complete)
249
- )
250
- # The run is closed out whatever breaks, backend setup and teardown included.
150
+ # The run is closed out whatever breaks, env setup and teardown included.
251
151
  try:
252
- async with backend as run_slot:
253
- # The display slots: in-process ones are the env's own (it fills their live
254
- # traces); a served rollout's is a client-side stand-in its worker never sees.
255
- planned = [
256
- slot
257
- for task, n in plan
258
- for slot in (
259
- env.slots(task, n)
260
- if env is not None
261
- else [RunSlot(task) for _ in range(n)]
262
- )
263
- ]
152
+ async with _in_process(env, config, semaphore, on_complete) as run_slot:
153
+ # the env's own slots: it fills their live traces as the rollouts run
154
+ planned = [slot for task, n in plan for slot in env.slots(task, n)]
264
155
  slots = [RunSlot.finished(episode) for episode in finished] + planned
265
156
  display = (
266
157
  dashboard(slots, config, start, push=push_state)
verifiers/v1/cli/gepa.py CHANGED
@@ -1,4 +1,4 @@
1
- """The GEPA entrypoint: `uv run gepa [<taskset-id>] --model <model> [options]`.
1
+ """The GEPA entrypoint: `uv run vf-gepa [<taskset-id>] --model <model> [options]`.
2
2
 
3
3
  Registered as the `gepa` console script. Optimizes a v1 taskset's `Task.system_prompt` via
4
4
  GEPA (Genetic-Pareto): alternating rollouts with a teacher LM reflecting on results — see
@@ -28,7 +28,7 @@ from verifiers.v1.utils.logging import setup_logging
28
28
 
29
29
  logger = logging.getLogger(__name__)
30
30
 
31
- USAGE = "usage: uv run gepa [<taskset-id>] [--env.id <id>] --model <model> [options] [@ file.toml]"
31
+ USAGE = "usage: uv run vf-gepa [<taskset-id>] [--env.id <id>] --model <model> [options] [@ file.toml]"
32
32
 
33
33
 
34
34
  def main(argv: list[str] | None = None) -> None:
verifiers/v1/cli/init.py CHANGED
@@ -8,7 +8,7 @@ from pydantic_config import cli
8
8
  from verifiers.v1.configs.cli.init import InitConfig
9
9
 
10
10
  USAGE = (
11
- "usage: uv run init <name> [--path ./environments] [-T/--add-tool] "
11
+ "usage: uv run vf-init <name> [--path ./environments] [-T/--add-tool] "
12
12
  "[-H/--add-harness]\n"
13
13
  " scaffold a new environment package"
14
14
  )
@@ -191,7 +191,7 @@ A v1 verifiers environment, scaffolded with `init`.
191
191
 
192
192
  ```bash
193
193
  uv pip install -e . # install this package (or register it in your project)
194
- uv run eval {dash} -n 3 # evaluate a few tasks with the bash harness
194
+ uv run vf-eval {dash} -n 3 # evaluate a few tasks with the bash harness
195
195
  ```
196
196
 
197
197
  ## Layout
@@ -228,7 +228,7 @@ def scaffold(config: InitConfig) -> Path:
228
228
  _write(pkg_dir / "servers" / "__init__.py", "")
229
229
  _write(pkg_dir / "servers" / "tool.py", _tool_py(stem, prefix))
230
230
 
231
- print(f"\ndone. next:\n uv pip install -e {env_dir}\n uv run eval {dash} -n 3")
231
+ print(f"\ndone. next:\n uv pip install -e {env_dir}\n uv run vf-eval {dash} -n 3")
232
232
  return env_dir
233
233
 
234
234
 
@@ -39,7 +39,7 @@ from verifiers.v1.utils.logging import setup_logging
39
39
  logger = logging.getLogger(__name__)
40
40
 
41
41
  USAGE = (
42
- "usage: uv run replay <output-dir> [options] [@ file.toml]\n"
42
+ "usage: uv run vf-replay <output-dir> [options] [@ file.toml]\n"
43
43
  " re-score a finished run's saved traces (judges + trace-only signals; no runtime)"
44
44
  )
45
45
 
@@ -49,9 +49,9 @@ REASONS = ("valid", "invalid", "unchecked", "error", "timeout")
49
49
  ResultRow = dict[str, Any]
50
50
 
51
51
  USAGE = (
52
- "usage: uv run validate [<taskset-id>] [--only-setup | --only-gold] "
52
+ "usage: uv run vf-validate [<taskset-id>] [--only-setup | --only-gold] "
53
53
  "[-o <output-dir>] [--runtime.type subprocess] [options] [@ file.toml]\n"
54
- " uv run validate @ <run-dir>/configs/validate.json --resume (re-run missing/errored/timed-out tasks)\n"
54
+ " uv run vf-validate @ <run-dir>/configs/validate.json --resume (re-run missing/errored/timed-out tasks)\n"
55
55
  " runs persisted gold and setup-only checks per task (no model)"
56
56
  )
57
57
 
@@ -9,7 +9,6 @@ from pydantic_config import BaseConfig
9
9
  from verifiers.v1.clients import ClientConfig, EvalClientConfig
10
10
  from verifiers.v1.configs.cli.env import narrowed_env_annotation, resolve_env_field
11
11
  from verifiers.v1.configs.env import EnvConfig
12
- from verifiers.v1.configs.serve import ServeConfig
13
12
  from verifiers.v1.envs.single_agent import SingleAgentEnvConfig
14
13
  from verifiers.v1.types import SamplingConfig
15
14
 
@@ -76,10 +75,6 @@ class EvalConfig(BaseConfig):
76
75
  env: SerializeAsAny[EnvConfig] = SingleAgentEnvConfig()
77
76
  """The environment — which env, its seed taskset, each agent, its knobs. Narrowed to
78
77
  the selected env's config class by the env id, else the taskset id."""
79
- serve: ServeConfig | None = Field(default_factory=ServeConfig)
80
- """How the env is hosted: the env-server worker pool (elastic by default) and each
81
- worker's episode bound — the path prime-rl trains through. `--no-serve` runs the
82
- rollouts in-process instead."""
83
78
  run: RunConfig = Field(default_factory=RunConfig)
84
79
  """Run identity: `run.name` is the display name, `run.dir` names the directory
85
80
  under `output_dir`, and `run.id` is stamped on traces."""
@@ -110,8 +105,7 @@ class EvalConfig(BaseConfig):
110
105
  )
111
106
  """Episodes in flight at once, `None` for no limit. An episode plays its agents one
112
107
  at a time, so this is the live agent runs too — until `--env.max-concurrent-agents`
113
- says otherwise. Under `[serve]` it also seeds each worker's bound, unless
114
- `--serve.max-concurrent` pins one."""
108
+ says otherwise."""
115
109
  verbose: bool = Field(False, validation_alias=AliasChoices("verbose", "v"))
116
110
  """Log at debug level instead of the default info."""
117
111
  dry_run: bool = Field(False, exclude=True)
@@ -122,8 +116,7 @@ class EvalConfig(BaseConfig):
122
116
  previous run's results. Excluded from the saved config."""
123
117
  rich: RichConfig | None = Field(default_factory=RichConfig)
124
118
  """The live dashboard (on by default; `--no-rich` streams logs to the console
125
- instead). A served run has no live per-turn view, so its rollout rows fill in as
126
- each episode completes; `--rich.show-logs` swaps the rows for the run's logs."""
119
+ instead); `--rich.show-logs` swaps the rollout rows for the run's logs."""
127
120
  push: bool = True
128
121
  """Upload the finished run to the Prime Intellect platform (the private Evaluations
129
122
  tab) at the end of the eval. On by default; disable with `--no-push`. Needs
@@ -136,7 +129,7 @@ class EvalConfig(BaseConfig):
136
129
  resume: bool = Field(False, exclude=True)
137
130
  """Re-run the run's missing/errored rollouts in place instead of starting fresh. The
138
131
  run dir comes from the resolved config (`output_dir / run.dir`), so resume with the
139
- run's own config — e.g. `uv run eval @ <run-dir>/configs/eval.json --resume`. Excluded
132
+ run's own config — e.g. `uv run vf-eval @ <run-dir>/configs/eval.json --resume`. Excluded
140
133
  from the saved config."""
141
134
 
142
135
  @model_validator(mode="before")
@@ -56,7 +56,7 @@ class ValidateConfig(BaseConfig):
56
56
  resume: bool = Field(False, exclude=True)
57
57
  """Re-run the run's missing, errored, and timed-out tasks in place. The run dir comes
58
58
  from the resolved config (`output_dir / run.dir`), so resume with the run's own
59
- config — e.g. `uv run validate @ <run-dir>/configs/validate.json --resume`.
59
+ config — e.g. `uv run vf-validate @ <run-dir>/configs/validate.json --resume`.
60
60
  Excluded from the saved config."""
61
61
  clean: bool = Field(False, exclude=True)
62
62
  """Delete the run directory (`output_dir / run.dir`) before running, overwriting a
@@ -4,7 +4,7 @@ A `BaseClientConfig` is an OpenAI-compatible endpoint (base_url + API-key env va
4
4
  + extra headers); `clients.resolve_client` turns one into a live `Client` — the
5
5
  interception server builds one per distinct config and shares it across the rollouts
6
6
  it multiplexes. The default Prime endpoint, API key, and team fall back to
7
- the active Prime CLI config, so direct `uv run eval` calls behave like `prime eval`.
7
+ the active Prime CLI config, so direct `uv run vf-eval` calls behave like `prime eval`.
8
8
  Both the eval entrypoint (its model client) and in-env LLM calls (e.g. a judge reward)
9
9
  build clients from these. `ClientConfig` is the CLI-selectable discriminated union
10
10
  (eval | train).
verifiers/v1/env.py CHANGED
@@ -316,17 +316,24 @@ class Env(ABC, Generic[ConfigT]):
316
316
  ctx: ModelContext,
317
317
  semaphore: asyncio.Semaphore | None = None,
318
318
  on_complete: Callable[[Episode], Awaitable[None]] | None = None,
319
+ on_trace: Callable[[Trace], None] | None = None,
319
320
  ) -> Episode:
320
321
  """Run one planned episode to completion, with whole-episode
321
322
  retries per `--env.retries`; `semaphore` bounds concurrent EPISODES — one
322
323
  permit for the attempt in flight, held across the whole of it (its agents,
323
324
  their boxes, `finalize()`) and released before a retry's backoff and before
324
- `on_complete` (the runners' persistence hook, which fires when final)."""
325
+ `on_complete` (the runners' persistence hook, which fires when final).
326
+ `on_trace` sees each trace at mint, after it joined `slot.traces`."""
325
327
 
326
328
  async def attempt() -> Episode:
327
329
  slot.traces = [] # a retry shows the fresh attempt's traces
328
330
  live = slot.traces
329
331
 
332
+ def minted(trace: Trace) -> None:
333
+ live.append(trace)
334
+ if on_trace is not None:
335
+ on_trace(trace)
336
+
330
337
  def discard(trace: Trace) -> None:
331
338
  # A retried agent attempt abandons its trace; drop it from the view.
332
339
  with contextlib.suppress(ValueError):
@@ -336,7 +343,7 @@ class Env(ABC, Generic[ConfigT]):
336
343
  return await self.run_episode(
337
344
  slot.task,
338
345
  ctx,
339
- on_trace=live.append,
346
+ on_trace=minted,
340
347
  on_discard=discard,
341
348
  )
342
349
 
verifiers/v1/graph.py CHANGED
@@ -548,8 +548,13 @@ class PendingTurn:
548
548
  """Add this turn to the graph; returns the committed assistant node's id."""
549
549
  assistant_id = _commit_turn(self, response)
550
550
  self.trace.tools = self.tools
551
+ self.trace.clear_preview(self)
551
552
  return assistant_id
552
553
 
554
+ def abandon(self) -> None:
555
+ """The request failed or was cancelled before a commit: its preview goes."""
556
+ self.trace.clear_preview(self)
557
+
553
558
  def commit_prompt(self) -> None:
554
559
  """Record an input that terminated before model inference."""
555
560
  parent = self.prefix_node_ids[-1] if self.prefix_node_ids else None
@@ -570,6 +575,7 @@ class PendingTurn:
570
575
  parent = len(self.trace.nodes) - 1
571
576
  index[_node_key(previous, message, self.tools)] = parent
572
577
  self.trace.tools = self.tools
578
+ self.trace.clear_preview(self)
573
579
 
574
580
 
575
581
  def prepare_turn(
@@ -483,6 +483,7 @@ class InterceptionServer(Interception):
483
483
  acp=acp,
484
484
  )
485
485
  )
486
+ session.trace.notify()
486
487
 
487
488
  async def handle_request(
488
489
  self, request: web.Request, dialect: Dialect
@@ -660,6 +661,9 @@ class InterceptionServer(Interception):
660
661
  return web.json_response(dialect.error_body(str(error)), status=400)
661
662
  except RolloutError as error:
662
663
  return self._fail(session, dialect, error)
664
+ # The tail is what the harness added since the last turn (tool results, user
665
+ # turns): live watchers see it now rather than with the model's reply.
666
+ session.trace.preview(turn, turn.tail)
663
667
 
664
668
  inspect_response = bool(session.response_interceptors or session.response_stops)
665
669
  if relay_streaming:
@@ -782,6 +786,8 @@ class InterceptionServer(Interception):
782
786
  error = e
783
787
  raise
784
788
  finally:
789
+ if node is None:
790
+ turn.abandon()
785
791
  # The turn's one per-exchange record: settings, timing, outcome, and
786
792
  # the error that ended it (if any).
787
793
  self.record_call(
@@ -1047,6 +1053,8 @@ class InterceptionServer(Interception):
1047
1053
  error = e
1048
1054
  raise
1049
1055
  finally:
1056
+ if node is None:
1057
+ turn.abandon()
1050
1058
  # The turn's one per-exchange record: settings, timing, outcome, and the
1051
1059
  # error that ended it (if any).
1052
1060
  self.record_call(
verifiers/v1/rollout.py CHANGED
@@ -180,6 +180,7 @@ class Rollout:
180
180
  proceed; a setup failure is captured onto the trace."""
181
181
  self._opened = True
182
182
  self.trace.timing.boot.start = time.time()
183
+ self.trace.notify()
183
184
  if self._borrowed_runtime is None:
184
185
  self.runtime = make_runtime(self.runtime_config, name=self.trace.id)
185
186
  elif self._borrowed_runtime is not None and self._borrowed_runtime.stopped:
@@ -219,6 +220,7 @@ class Rollout:
219
220
  now = time.time()
220
221
  self.trace.timing.boot.end = now
221
222
  self.trace.timing.setup.start = now
223
+ self.trace.notify()
222
224
  # Task setup and harness provisioning share one setup-stage deadline.
223
225
  setup_deadline = (
224
226
  None
@@ -342,6 +344,7 @@ class Rollout:
342
344
  now = time.time()
343
345
  self.trace.timing.setup.end = now
344
346
  self.trace.timing.agent.start = now
347
+ self.trace.notify()
345
348
  return not self._session.stopped
346
349
 
347
350
  async def step(self, messages: Messages | None = None) -> bool:
@@ -464,6 +467,7 @@ class Rollout:
464
467
  finally:
465
468
  if trace.timing.agent.start and not trace.timing.agent.end:
466
469
  trace.timing.agent.end = time.time()
470
+ trace.notify()
467
471
  if not self._failed and self._opened:
468
472
  assert runtime is not None
469
473
  trace.timing.finalize.start = time.time()
@@ -479,6 +483,7 @@ class Rollout:
479
483
  now = time.time()
480
484
  trace.timing.finalize.end = now
481
485
  trace.timing.scoring.start = now
486
+ trace.notify()
482
487
  async with boundary(TaskError, "scoring"):
483
488
  # Cross-trace judgement runs later, after the runtime is gone.
484
489
  await asyncio.wait_for(
@@ -489,6 +494,7 @@ class Rollout:
489
494
  self._timeouts.scoring,
490
495
  )
491
496
  trace.timing.scoring.end = time.time()
497
+ trace.notify()
492
498
  except Exception as e: # noqa: BLE001 - finalize boundary records every rollout failure
493
499
  self.fail(e)
494
500
  finally:
@@ -1,4 +1,5 @@
1
1
  from verifiers.v1.serve.client import EnvClient
2
+ from verifiers.v1.serve.delta import EpisodeAssembly
2
3
  from verifiers.v1.serve.pool import EnvServerPool, env_config_data, serve_env
3
4
  from verifiers.v1.serve.server import EnvServer
4
5
  from verifiers.v1.serve.types import (
@@ -12,6 +13,7 @@ __all__ = [
12
13
  "EnvClient",
13
14
  "EnvServer",
14
15
  "EnvServerPool",
16
+ "EpisodeAssembly",
15
17
  "HealthRequest",
16
18
  "HealthResponse",
17
19
  "RunRequest",