verifiers 0.2.2.dev48__py3-none-any.whl → 0.2.2.dev50__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/agent.py +12 -12
- verifiers/v1/cli/dashboard/eval.py +13 -16
- verifiers/v1/cli/debug.py +2 -2
- verifiers/v1/cli/eval/runner.py +3 -3
- verifiers/v1/cli/replay.py +1 -1
- verifiers/v1/cli/validate.py +12 -0
- verifiers/v1/configs/judge.py +3 -10
- verifiers/v1/configs/taskset.py +5 -0
- verifiers/v1/env.py +1 -1
- verifiers/v1/envs/agentic_judge/env.py +12 -22
- verifiers/v1/envs/user_sim/env.py +2 -2
- verifiers/v1/errors.py +1 -1
- verifiers/v1/gepa/adapter.py +6 -11
- verifiers/v1/gepa/runner.py +14 -1
- verifiers/v1/judge.py +4 -4
- verifiers/v1/legacy.py +6 -2
- verifiers/v1/push.py +5 -5
- verifiers/v1/retries.py +3 -1
- verifiers/v1/rollout.py +23 -20
- verifiers/v1/runtimes/subprocess.py +1 -1
- verifiers/v1/task.py +12 -2
- verifiers/v1/taskset.py +8 -3
- verifiers/v1/tasksets/harbor/taskset.py +4 -4
- verifiers/v1/tasksets/lean/taskset.py +1 -2
- verifiers/v1/trace.py +135 -238
- verifiers/v1/utils/compile.py +8 -8
- {verifiers-0.2.2.dev48.dist-info → verifiers-0.2.2.dev50.dist-info}/METADATA +1 -1
- {verifiers-0.2.2.dev48.dist-info → verifiers-0.2.2.dev50.dist-info}/RECORD +31 -31
- {verifiers-0.2.2.dev48.dist-info → verifiers-0.2.2.dev50.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev48.dist-info → verifiers-0.2.2.dev50.dist-info}/entry_points.txt +0 -0
- {verifiers-0.2.2.dev48.dist-info → verifiers-0.2.2.dev50.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/agent.py
CHANGED
|
@@ -45,7 +45,7 @@ from verifiers.v1.types import (
|
|
|
45
45
|
UserMessage,
|
|
46
46
|
)
|
|
47
47
|
from verifiers.v1.utils.compile import (
|
|
48
|
-
|
|
48
|
+
cap_remote_agent_timeout,
|
|
49
49
|
resolve_runtime_config,
|
|
50
50
|
validate_pairing,
|
|
51
51
|
)
|
|
@@ -390,7 +390,7 @@ class Agent:
|
|
|
390
390
|
attempt + 1,
|
|
391
391
|
retry.max_retries,
|
|
392
392
|
delay,
|
|
393
|
-
trace.
|
|
393
|
+
trace.last_error.type if trace.last_error else "?",
|
|
394
394
|
)
|
|
395
395
|
await asyncio.sleep(delay)
|
|
396
396
|
if history:
|
|
@@ -419,8 +419,8 @@ class Agent:
|
|
|
419
419
|
# close() never runs — free the run's servers and owned runtime first.
|
|
420
420
|
await run.abort()
|
|
421
421
|
raise
|
|
422
|
-
if trace.runtime is not None:
|
|
423
|
-
trace.runtime.borrowed = runtime is not None
|
|
422
|
+
if trace.agent.runtime is not None:
|
|
423
|
+
trace.agent.runtime.borrowed = runtime is not None
|
|
424
424
|
return trace
|
|
425
425
|
|
|
426
426
|
@asynccontextmanager
|
|
@@ -482,8 +482,8 @@ class Agent:
|
|
|
482
482
|
opened = await run.open()
|
|
483
483
|
if not opened:
|
|
484
484
|
trace = await run.close()
|
|
485
|
-
if trace.runtime is not None:
|
|
486
|
-
trace.runtime.borrowed = runtime is not None
|
|
485
|
+
if trace.agent.runtime is not None:
|
|
486
|
+
trace.agent.runtime.borrowed = runtime is not None
|
|
487
487
|
if not opened:
|
|
488
488
|
failure = run.failure
|
|
489
489
|
if failure is None: # `open()` returning False always captures one.
|
|
@@ -499,8 +499,8 @@ class Agent:
|
|
|
499
499
|
raise
|
|
500
500
|
finally:
|
|
501
501
|
trace = run.trace if run.closed else await interaction.close()
|
|
502
|
-
if trace.runtime is not None:
|
|
503
|
-
trace.runtime.borrowed = runtime is not None
|
|
502
|
+
if trace.agent.runtime is not None:
|
|
503
|
+
trace.agent.runtime.borrowed = runtime is not None
|
|
504
504
|
|
|
505
505
|
def _rollout_params(
|
|
506
506
|
self, task: Task, runtime: Runtime | None, shared_tools: dict
|
|
@@ -520,10 +520,10 @@ class Agent:
|
|
|
520
520
|
self.harness, type(task), runtime_config, shared_tools=shared_tools
|
|
521
521
|
)
|
|
522
522
|
# Timeout precedence: agent-level wins, else the task's, else no limit.
|
|
523
|
-
|
|
523
|
+
agent_timeout = (
|
|
524
524
|
self.timeout.rollout
|
|
525
525
|
if self.timeout.rollout is not None
|
|
526
|
-
else task.data.timeout.
|
|
526
|
+
else task.data.timeout.agent
|
|
527
527
|
)
|
|
528
528
|
return {
|
|
529
529
|
"agent_config": self.config,
|
|
@@ -535,8 +535,8 @@ class Agent:
|
|
|
535
535
|
if self.timeout.setup is not None
|
|
536
536
|
else task.data.timeout.setup
|
|
537
537
|
),
|
|
538
|
-
"
|
|
539
|
-
|
|
538
|
+
"agent_timeout": cap_remote_agent_timeout(
|
|
539
|
+
agent_timeout, runtime_config, task
|
|
540
540
|
),
|
|
541
541
|
"finalize_timeout": (
|
|
542
542
|
self.timeout.finalize
|
|
@@ -302,7 +302,7 @@ def Progress(
|
|
|
302
302
|
# run, a modeled user) are `trainable=False` and carry no rewards, so counting
|
|
303
303
|
# them dilutes every mean with structural zeros. An all-untrainable run (every
|
|
304
304
|
# role frozen) falls back to all traces rather than showing nothing.
|
|
305
|
-
scored = [t for t in done_traces if t.trainable] or done_traces
|
|
305
|
+
scored = [t for t in done_traces if t.agent.trainable] or done_traces
|
|
306
306
|
total = len(slots)
|
|
307
307
|
# Headline reward = mean over non-errored traces; when any errored, `format_mean` appends
|
|
308
308
|
# the global avg (errored count as 0) in parens. `err` is the share of episodes that
|
|
@@ -367,13 +367,13 @@ def _breakdown(scored: list[Trace], done: list[Trace]) -> Table | None:
|
|
|
367
367
|
# show); usage/time below still cover errored rollouts (their resources were spent regardless).
|
|
368
368
|
has_clean = any(not t.has_error for t in done)
|
|
369
369
|
score_rows = (("rewards", "rewards"), ("metrics", "metrics")) if has_clean else ()
|
|
370
|
-
by_agent: dict[str
|
|
370
|
+
by_agent: dict[str, list[Trace]] = {}
|
|
371
371
|
for trace in done:
|
|
372
|
-
by_agent.setdefault(trace.
|
|
372
|
+
by_agent.setdefault(trace.agent.name, []).append(trace)
|
|
373
373
|
for label, source in score_rows:
|
|
374
374
|
if len(by_agent) > 1:
|
|
375
375
|
segments = [
|
|
376
|
-
f"[dim]{name
|
|
376
|
+
f"[dim]{name}:[/dim] {means}"
|
|
377
377
|
for name, traces in by_agent.items()
|
|
378
378
|
if (means := _score_segments(traces, source)) is not None
|
|
379
379
|
]
|
|
@@ -505,7 +505,7 @@ def _stage(trace: Trace) -> str:
|
|
|
505
505
|
stage = "boot" # trace minted, first span not yet opened (an instant)
|
|
506
506
|
# A boot stuck on a first-use platform image build reads differently from a
|
|
507
507
|
# normal boot — it can sit there for ~10 minutes (prime runtime only).
|
|
508
|
-
if stage == "boot" and getattr(trace.runtime, "image_cached", None) is False:
|
|
508
|
+
if stage == "boot" and getattr(trace.agent.runtime, "image_cached", None) is False:
|
|
509
509
|
return "build"
|
|
510
510
|
return stage
|
|
511
511
|
|
|
@@ -564,14 +564,14 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
|
|
|
564
564
|
group_rows.append(("pending", [f"task {base}", *[""] * 7], "", ""))
|
|
565
565
|
continue
|
|
566
566
|
for t in slot.traces:
|
|
567
|
-
label = f"{base} agent={t.
|
|
567
|
+
label = f"{base} agent={t.agent.name}"
|
|
568
568
|
if slot.done: # fully scored — reward is final
|
|
569
569
|
state = "error" if t.has_error else "success"
|
|
570
570
|
# A trace that recorded nothing shows no reward: a judge or
|
|
571
571
|
# modeled-user seat's `reward=0.00` would read as a score.
|
|
572
572
|
result = (
|
|
573
|
-
t.
|
|
574
|
-
if t.has_error
|
|
573
|
+
t.last_error.type
|
|
574
|
+
if t.has_error and t.last_error
|
|
575
575
|
else (f"reward={t.reward:.2f}" if t.rewards else "")
|
|
576
576
|
)
|
|
577
577
|
if t.has_error:
|
|
@@ -582,7 +582,7 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
|
|
|
582
582
|
t.is_truncated
|
|
583
583
|
): # flag a clipped rollout next to its stop condition
|
|
584
584
|
stop = f"{stop} (truncated)".strip()
|
|
585
|
-
elif t.is_completed and (err := t.
|
|
585
|
+
elif t.is_completed and (err := t.last_error) is not None:
|
|
586
586
|
# An errored trace whose episode is still running its other
|
|
587
587
|
# traces (or `score()`) is already a failure — show it, don't
|
|
588
588
|
# let it sit as "scoring" until the whole episode lands.
|
|
@@ -592,12 +592,9 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
|
|
|
592
592
|
# The trace's own stamp, not the run-level runtime: a role's harness
|
|
593
593
|
# may resolve elsewhere (the judge env's sandboxed judge on a
|
|
594
594
|
# subprocess run).
|
|
595
|
-
if t.runtime is not None:
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
if t.runtime.id
|
|
599
|
-
else t.runtime.type
|
|
600
|
-
)
|
|
595
|
+
if t.agent.runtime is not None:
|
|
596
|
+
rt = t.agent.runtime
|
|
597
|
+
runtime = f"{rt.type}({rt.id})" if rt.id else rt.type
|
|
601
598
|
else:
|
|
602
599
|
runtime = runtime_type
|
|
603
600
|
turns = t.num_turns
|
|
@@ -635,7 +632,7 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
|
|
|
635
632
|
f"{nbranches} branch{'es' * (nbranches != 1)}",
|
|
636
633
|
tokens,
|
|
637
634
|
f"{format_cost_usd(cost)}" if cost is not None else "",
|
|
638
|
-
stop, # stop condition (agent_completed / max_turns /
|
|
635
|
+
stop, # stop condition (agent_completed / max_turns / error), once done
|
|
639
636
|
]
|
|
640
637
|
# No start time yet (queued, not generating) → blank, not `now - 0` (~56 years).
|
|
641
638
|
elapsed = format_time(end - start) if start else ""
|
verifiers/v1/cli/debug.py
CHANGED
|
@@ -101,9 +101,9 @@ def error_info(
|
|
|
101
101
|
|
|
102
102
|
|
|
103
103
|
def capture_trace_error(trace: Trace, error: BaseException) -> None:
|
|
104
|
-
# CancelledError is a BaseException; Trace.
|
|
104
|
+
# CancelledError is a BaseException; Trace.record_error accepts Exception.
|
|
105
105
|
if isinstance(error, Exception):
|
|
106
|
-
trace.
|
|
106
|
+
trace.record_error(error)
|
|
107
107
|
return
|
|
108
108
|
trace.errors.append(
|
|
109
109
|
Error(
|
verifiers/v1/cli/eval/runner.py
CHANGED
|
@@ -68,7 +68,7 @@ async def run_eval(env: Env, config: EvalConfig) -> list[Episode]:
|
|
|
68
68
|
|
|
69
69
|
async def on_complete(episode: Episode) -> None:
|
|
70
70
|
for trace in episode.traces:
|
|
71
|
-
trace.
|
|
71
|
+
trace.record_run(EvalRunInfo(id=config.uuid))
|
|
72
72
|
await append_episode(out, episode, write_lock)
|
|
73
73
|
|
|
74
74
|
# Serving resources (shared tool servers, interception) come up once for the
|
|
@@ -240,7 +240,7 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
|
|
|
240
240
|
)
|
|
241
241
|
records = []
|
|
242
242
|
for trace in traces:
|
|
243
|
-
trace.
|
|
243
|
+
trace.record_run(EvalRunInfo(id=config.uuid))
|
|
244
244
|
await append_trace(out, trace, write_lock, env=config.env_id)
|
|
245
245
|
records.append(Episode.of(trace))
|
|
246
246
|
return records
|
|
@@ -254,7 +254,7 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
|
|
|
254
254
|
**payload,
|
|
255
255
|
)
|
|
256
256
|
for trace in episode.traces:
|
|
257
|
-
trace.
|
|
257
|
+
trace.record_run(EvalRunInfo(id=config.uuid))
|
|
258
258
|
await append_episode(out, episode, write_lock)
|
|
259
259
|
return [episode]
|
|
260
260
|
|
verifiers/v1/cli/replay.py
CHANGED
|
@@ -165,7 +165,7 @@ async def run_replay(config: ReplayConfig, source: Path, out: Path) -> list[Trac
|
|
|
165
165
|
st.state, st.detail = "scored", f"reward {trace.reward:.3f}"
|
|
166
166
|
except Exception as exc:
|
|
167
167
|
st.state, st.detail = "error", type(exc).__name__
|
|
168
|
-
trace.
|
|
168
|
+
trace.record_error(exc)
|
|
169
169
|
if not config.rich:
|
|
170
170
|
logger.warning(
|
|
171
171
|
"replay: scoring failed for task %s",
|
verifiers/v1/cli/validate.py
CHANGED
|
@@ -89,6 +89,12 @@ async def _run_gold(task: Task, config: ValidateConfig) -> ResultRow:
|
|
|
89
89
|
trace = Trace(
|
|
90
90
|
task=TraceTask(type=type(task).__name__, data=task.data),
|
|
91
91
|
state=state_cls(type(task))(),
|
|
92
|
+
# No agent runs here — the info only records the runtime policy.
|
|
93
|
+
agent=vf.AgentInfo(
|
|
94
|
+
config=vf.AgentConfig(runtime=config.runtime),
|
|
95
|
+
name="validate",
|
|
96
|
+
trainable=False,
|
|
97
|
+
),
|
|
92
98
|
)
|
|
93
99
|
await runtime.start()
|
|
94
100
|
await asyncio.wait_for(
|
|
@@ -124,6 +130,12 @@ async def _run_setup(task: Task, config: ValidateConfig) -> ResultRow:
|
|
|
124
130
|
trace = Trace(
|
|
125
131
|
task=TraceTask(type=type(task).__name__, data=task.data),
|
|
126
132
|
state=state_cls(type(task))(),
|
|
133
|
+
# No agent runs here — the info only records the runtime policy.
|
|
134
|
+
agent=vf.AgentInfo(
|
|
135
|
+
config=vf.AgentConfig(runtime=config.runtime),
|
|
136
|
+
name="validate",
|
|
137
|
+
trainable=False,
|
|
138
|
+
),
|
|
127
139
|
)
|
|
128
140
|
await runtime.start()
|
|
129
141
|
await asyncio.wait_for(
|
verifiers/v1/configs/judge.py
CHANGED
|
@@ -5,7 +5,7 @@ from collections.abc import Sequence
|
|
|
5
5
|
from pathlib import Path
|
|
6
6
|
from typing import Any
|
|
7
7
|
|
|
8
|
-
from pydantic import BaseModel, SerializeAsAny
|
|
8
|
+
from pydantic import BaseModel, SerializeAsAny
|
|
9
9
|
|
|
10
10
|
from verifiers.v1.clients import BaseClientConfig
|
|
11
11
|
from verifiers.v1.types import ID, SamplingConfig
|
|
@@ -20,15 +20,8 @@ class JudgeConfig(BaseClientConfig):
|
|
|
20
20
|
weight: float = 1.0
|
|
21
21
|
model: str = "openai/gpt-5.4-nano"
|
|
22
22
|
sampling: SamplingConfig = SamplingConfig()
|
|
23
|
-
prompt:
|
|
24
|
-
|
|
25
|
-
"""Prompt file override, mutually exclusive with `prompt`."""
|
|
26
|
-
|
|
27
|
-
@model_validator(mode="after")
|
|
28
|
-
def check_prompt_source(self) -> "JudgeConfig":
|
|
29
|
-
if self.prompt is not None and self.prompt_file is not None:
|
|
30
|
-
raise ValueError("set `prompt` or `prompt_file`, not both")
|
|
31
|
-
return self
|
|
23
|
+
prompt: Path | None = None
|
|
24
|
+
"""File whose text overrides the judge's default prompt template."""
|
|
32
25
|
|
|
33
26
|
|
|
34
27
|
Judges = list[SerializeAsAny[JudgeConfig]]
|
verifiers/v1/configs/taskset.py
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
"""The taskset plugin's config: which rows load, under `--env.taskset.*`."""
|
|
2
2
|
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
3
5
|
from pydantic import SerializeAsAny
|
|
4
6
|
from pydantic_config import BaseConfig
|
|
5
7
|
|
|
@@ -14,6 +16,9 @@ class TasksetConfig(BaseConfig):
|
|
|
14
16
|
positional `eval <taskset-id>`)."""
|
|
15
17
|
task: SerializeAsAny[TaskConfig] = TaskConfig()
|
|
16
18
|
"""Config passed to each task, under `--env.taskset.task.*`."""
|
|
19
|
+
system_prompt: Path | None = None
|
|
20
|
+
"""File whose text overrides each task's `TaskData.system_prompt` in
|
|
21
|
+
`Taskset.select` (e.g. a GEPA `best_system_prompt.txt`)."""
|
|
17
22
|
|
|
18
23
|
@property
|
|
19
24
|
def name(self) -> str:
|
verifiers/v1/env.py
CHANGED
|
@@ -163,7 +163,7 @@ class Env(ABC, Generic[ConfigT]):
|
|
|
163
163
|
"""Cross-agent judgement — THE programmable judgement surface: plain
|
|
164
164
|
imperative Python over the finished episode (per-trace judgement already
|
|
165
165
|
ran on each trace's own task). `episode.traces` is the flat episode in
|
|
166
|
-
completion order, each trace's `
|
|
166
|
+
completion order, each trace's `agent.name` stamp naming its agent; attach
|
|
167
167
|
signals via `record_reward`/`record_metric`, in program order. A raise
|
|
168
168
|
fails the episode (the retryable unit) — validate strictly, never
|
|
169
169
|
record a guess."""
|
|
@@ -180,41 +180,31 @@ class JudgeTask(vf.Task):
|
|
|
180
180
|
class JudgeTaskConfig(vf.BaseConfig):
|
|
181
181
|
"""The judge's minted task: the grading policy and what lands in its box."""
|
|
182
182
|
|
|
183
|
-
prompt: Path |
|
|
184
|
-
"""Grading-policy
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
`.txt` file): task-family pointers into the trace or box — e.g. for math,
|
|
192
|
-
where the reference answer lives in the record; for SWE, to diff the repo
|
|
193
|
-
or read `info.patch`."""
|
|
183
|
+
prompt: Path | None = None
|
|
184
|
+
"""Grading-policy file. Replaces only the policy body — the verdict contract
|
|
185
|
+
and workspace note are always appended. May reference `{prompt}` (the solver
|
|
186
|
+
task's prompt); if it doesn't, the task statement is appended after."""
|
|
187
|
+
hint: Path | None = None
|
|
188
|
+
"""Optional hints file injected as their own section: task-family pointers
|
|
189
|
+
into the trace or box — e.g. for math, where the reference answer lives in
|
|
190
|
+
the record; for SWE, to diff the repo or read `info.patch`."""
|
|
194
191
|
rubric: Path | None = None
|
|
195
192
|
"""Criteria the judge grades against: a `.toml`/`.json` file with a
|
|
196
193
|
`criteria` list — the plugged rubric judge's format, so the same rubric
|
|
197
194
|
files work for both. None grades the single built-in `solved` criterion."""
|
|
198
195
|
|
|
199
|
-
@staticmethod
|
|
200
|
-
def _resolve(value: Path | str) -> str:
|
|
201
|
-
path = Path(value)
|
|
202
|
-
if isinstance(value, Path) or path.suffix in (".md", ".txt"):
|
|
203
|
-
return path.read_text(encoding="utf-8")
|
|
204
|
-
return str(value)
|
|
205
|
-
|
|
206
196
|
def build_prompt(self) -> str:
|
|
207
197
|
if self.prompt is None:
|
|
208
198
|
return GRADE_PROMPT + "\n\n" + TASK_SECTION
|
|
209
|
-
return self.
|
|
199
|
+
return self.prompt.read_text()
|
|
210
200
|
|
|
211
201
|
def build_hint(self) -> str | None:
|
|
212
|
-
return self.
|
|
202
|
+
return self.hint.read_text() if self.hint is not None else None
|
|
213
203
|
|
|
214
204
|
def criteria(self) -> list[Criterion]:
|
|
215
205
|
if self.rubric is None:
|
|
216
206
|
return [SOLVED]
|
|
217
|
-
text = self.rubric.read_text(
|
|
207
|
+
text = self.rubric.read_text()
|
|
218
208
|
data = (
|
|
219
209
|
tomllib.loads(text)
|
|
220
210
|
if self.rubric.suffix.lower() == ".toml"
|
|
@@ -304,7 +294,7 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
|
|
|
304
294
|
await agents.judge.run(judge_task, runtime=box)
|
|
305
295
|
|
|
306
296
|
async def finalize(self, task: vf.Task, episode: vf.Episode) -> None:
|
|
307
|
-
by_agent = {t.
|
|
297
|
+
by_agent = {t.agent.name: t for t in episode.traces}
|
|
308
298
|
solution, verdict = by_agent["solver"], by_agent["judge"]
|
|
309
299
|
data = verdict.info.get("verdict")
|
|
310
300
|
if not isinstance(data, dict) or not isinstance(data.get("verdicts"), list):
|
|
@@ -89,7 +89,7 @@ class UserSimEnv(vf.Env[UserSimEnvConfig]):
|
|
|
89
89
|
async def finalize(self, task, episode):
|
|
90
90
|
"""One conversation-shape fact about the user's side, recorded on the
|
|
91
91
|
assistant's trace; judgement stays on the task's rewards."""
|
|
92
|
-
(user,) = (t for t in episode.traces if t.
|
|
92
|
+
(user,) = (t for t in episode.traces if t.agent.name == "user")
|
|
93
93
|
for trace in episode.traces:
|
|
94
|
-
if trace.
|
|
94
|
+
if trace.agent.name == "assistant":
|
|
95
95
|
trace.record_metric("user_turns", float(user.num_turns))
|
verifiers/v1/errors.py
CHANGED
|
@@ -4,7 +4,7 @@ Four mechanisms, each in one place:
|
|
|
4
4
|
|
|
5
5
|
1. Vocabulary (this module): `RolloutError` and the flat boundary types below. Each names the
|
|
6
6
|
boundary a failure crossed — provider, harness, toolset, sandbox, task, or
|
|
7
|
-
interception — so a recorded `trace.
|
|
7
|
+
interception — so a recorded `trace.last_error.type` says where the rollout broke.
|
|
8
8
|
2. Classification (`boundary`): the one helper that runs a framework→code boundary and attributes
|
|
9
9
|
any escaping error to that boundary's type. Extension code (task hooks, harness subclasses)
|
|
10
10
|
raises plain Python errors — it never constructs a `vf` error type; `boundary` classifies them.
|
verifiers/v1/gepa/adapter.py
CHANGED
|
@@ -68,14 +68,10 @@ class GEPAAdapter:
|
|
|
68
68
|
)
|
|
69
69
|
|
|
70
70
|
async def _run_batch(self, batch: list[int], system_prompt: str) -> list[Episode]:
|
|
71
|
-
# Inject the candidate
|
|
72
|
-
#
|
|
73
|
-
tasks
|
|
74
|
-
|
|
75
|
-
t.data.model_copy(update={"system_prompt": system_prompt}), t.config
|
|
76
|
-
)
|
|
77
|
-
for t in (self.tasks[idx] for idx in batch)
|
|
78
|
-
]
|
|
71
|
+
# Inject the candidate as a copy of each base task with its system_prompt overridden
|
|
72
|
+
# (`with_system_prompt` copies rather than reconstructs, so subclass state survives and
|
|
73
|
+
# the shared base task in `self.tasks` is left untouched for the next candidate).
|
|
74
|
+
tasks = [self.tasks[idx].with_system_prompt(system_prompt) for idx in batch]
|
|
79
75
|
slots = [slot for task in tasks for slot in self.env.slots(task)]
|
|
80
76
|
results = await asyncio.gather(
|
|
81
77
|
*(
|
|
@@ -104,10 +100,9 @@ class GEPAAdapter:
|
|
|
104
100
|
"completion": trace.last_reply,
|
|
105
101
|
"reward": trace.reward,
|
|
106
102
|
}
|
|
107
|
-
|
|
108
|
-
record["agent"] = trace.agent_name
|
|
103
|
+
record["agent"] = trace.agent.name
|
|
109
104
|
if trace.has_error:
|
|
110
|
-
record["error"] = str(trace.
|
|
105
|
+
record["error"] = str(trace.last_error)
|
|
111
106
|
if trace.stop_condition:
|
|
112
107
|
record["stop_condition"] = trace.stop_condition
|
|
113
108
|
for column in self.reflection_columns:
|
verifiers/v1/gepa/runner.py
CHANGED
|
@@ -105,7 +105,20 @@ def run_gepa(env: Env, config: GEPAConfig) -> GEPAResult:
|
|
|
105
105
|
"skip_perfect_score": False,
|
|
106
106
|
"logger": _GEPALog(),
|
|
107
107
|
}
|
|
108
|
-
|
|
108
|
+
result = optimize(**optimize_kwargs)
|
|
109
|
+
if run_dir is not None:
|
|
110
|
+
# Persist the winning prompt as a plain file so it can be handed straight to
|
|
111
|
+
# eval/train via `--env.taskset.system-prompt` (see TasksetConfig).
|
|
112
|
+
candidate = result.best_candidate
|
|
113
|
+
best = (
|
|
114
|
+
candidate.get("system_prompt", "")
|
|
115
|
+
if isinstance(candidate, dict)
|
|
116
|
+
else str(candidate)
|
|
117
|
+
)
|
|
118
|
+
best_path = run_dir / "best_system_prompt.txt"
|
|
119
|
+
best_path.write_text(best, encoding="utf-8")
|
|
120
|
+
logger.info("best system prompt: %s", best_path)
|
|
121
|
+
return result
|
|
109
122
|
finally:
|
|
110
123
|
loop.run_until_complete(serving.__aexit__(None, None, None))
|
|
111
124
|
finally:
|
verifiers/v1/judge.py
CHANGED
|
@@ -116,13 +116,13 @@ def judge_config_cls(cls: type) -> type[JudgeConfig]:
|
|
|
116
116
|
|
|
117
117
|
class Judge(Generic[ParsedT, ConfigT]):
|
|
118
118
|
prompt: str | None = None
|
|
119
|
-
"""Default prompt template, overridden by config."""
|
|
119
|
+
"""Default prompt template, overridden by a config `prompt` file."""
|
|
120
120
|
schema: type[BaseModel] | None = None
|
|
121
121
|
|
|
122
122
|
def __init__(self, config: ConfigT | None = None) -> None:
|
|
123
123
|
self.config = cast(ConfigT, config or judge_config_cls(type(self))())
|
|
124
|
-
if self.config.
|
|
125
|
-
self.prompt = self.config.
|
|
124
|
+
if self.config.prompt is not None:
|
|
125
|
+
self.prompt = self.config.prompt.read_text()
|
|
126
126
|
|
|
127
127
|
@property
|
|
128
128
|
def reward_name(self) -> str:
|
|
@@ -132,7 +132,7 @@ class Judge(Generic[ParsedT, ConfigT]):
|
|
|
132
132
|
return judge_key(self.config) or fallback or "judge"
|
|
133
133
|
|
|
134
134
|
def build_messages(self, **fields: Any) -> str | Messages:
|
|
135
|
-
template = self.
|
|
135
|
+
template = self.prompt
|
|
136
136
|
if template is None:
|
|
137
137
|
raise ValueError(
|
|
138
138
|
f"{type(self).__name__} has no `prompt`; set it or override build_messages"
|
verifiers/v1/legacy.py
CHANGED
|
@@ -24,6 +24,7 @@ from pydantic import ValidationError
|
|
|
24
24
|
|
|
25
25
|
from verifiers.v1 import graph
|
|
26
26
|
from verifiers.v1.clients.config import ClientConfig, TrainClientConfig
|
|
27
|
+
from verifiers.v1.configs.agent import AgentConfig
|
|
27
28
|
from verifiers.v1.episode import Episode
|
|
28
29
|
from verifiers.v1.serve.server import EnvServer
|
|
29
30
|
from verifiers.v1.serve.types import (
|
|
@@ -34,6 +35,7 @@ from verifiers.v1.serve.types import (
|
|
|
34
35
|
)
|
|
35
36
|
from verifiers.v1.task import WireTaskData
|
|
36
37
|
from verifiers.v1.trace import (
|
|
38
|
+
AgentInfo,
|
|
37
39
|
Error,
|
|
38
40
|
GenerationSpan,
|
|
39
41
|
ModelCall,
|
|
@@ -227,7 +229,6 @@ def _timing(raw: Any) -> Timing:
|
|
|
227
229
|
_V0_TO_V1_TRUNCATION_STOP = {
|
|
228
230
|
"max_turns_reached": "max_turns",
|
|
229
231
|
"prompt_too_long": "context_length",
|
|
230
|
-
"timeout_reached": "harness_timeout",
|
|
231
232
|
"max_total_completion_tokens_reached": "max_output_tokens",
|
|
232
233
|
}
|
|
233
234
|
|
|
@@ -270,7 +271,10 @@ def rollout_output_to_trace(out: dict, task_idx: int) -> Trace:
|
|
|
270
271
|
type="Task",
|
|
271
272
|
data=_to_wire_task(task_idx, out.get("prompt"), out.get("answer")),
|
|
272
273
|
),
|
|
273
|
-
|
|
274
|
+
# v0 rollouts carry no agent config — record the default, like the
|
|
275
|
+
# base task type above.
|
|
276
|
+
agent=AgentInfo(config=AgentConfig()),
|
|
277
|
+
tools=_to_v1_tools(out.get("tool_defs")) or [],
|
|
274
278
|
rewards={"reward": Reward(score=float(out.get("reward") or 0.0))},
|
|
275
279
|
metrics={k: float(v) for k, v in (out.get("metrics") or {}).items()},
|
|
276
280
|
info=dict(out.get("info") or {}),
|
verifiers/v1/push.py
CHANGED
|
@@ -58,8 +58,8 @@ def trace_to_sample(
|
|
|
58
58
|
"example_id": trace.task.data.idx,
|
|
59
59
|
"rollout_number": rollout_number,
|
|
60
60
|
"episode_id": episode_id,
|
|
61
|
-
"agent": trace.
|
|
62
|
-
"trainable": trace.trainable,
|
|
61
|
+
"agent": trace.agent.name,
|
|
62
|
+
"trainable": trace.agent.trainable,
|
|
63
63
|
"task": task,
|
|
64
64
|
"prompt": [],
|
|
65
65
|
"completion": dump(branches[-1].messages) if branches else [],
|
|
@@ -73,8 +73,8 @@ def trace_to_sample(
|
|
|
73
73
|
"is_completed": trace.is_completed,
|
|
74
74
|
"is_truncated": trace.is_truncated,
|
|
75
75
|
"metrics": trace.metrics,
|
|
76
|
-
"error": trace.
|
|
77
|
-
if trace.
|
|
76
|
+
"error": trace.last_error.model_dump(mode="json", exclude_none=True)
|
|
77
|
+
if trace.last_error
|
|
78
78
|
else None,
|
|
79
79
|
"stop_condition": trace.stop_condition,
|
|
80
80
|
"trajectory": [
|
|
@@ -124,7 +124,7 @@ def _run_metrics(episodes: list[Episode], traces: list[Trace]) -> dict[str, Any]
|
|
|
124
124
|
back to all traces when none are trainable (same rule as the dashboard).
|
|
125
125
|
`avg_error` is the share of EPISODES that aren't ok: a hook failure counts
|
|
126
126
|
even when its traces are clean or it left none."""
|
|
127
|
-
scored = [t for t in traces if t.trainable] or traces
|
|
127
|
+
scored = [t for t in traces if t.agent.trainable] or traces
|
|
128
128
|
sums: dict[str, float] = {}
|
|
129
129
|
counts: dict[str, int] = {}
|
|
130
130
|
for trace in scored:
|
verifiers/v1/retries.py
CHANGED
|
@@ -117,7 +117,9 @@ async def run_episode_with_retry(
|
|
|
117
117
|
final = await run()
|
|
118
118
|
if attempt == retry.max_retries or not episode_should_retry(final, retry):
|
|
119
119
|
break
|
|
120
|
-
cause = final.error or next(
|
|
120
|
+
cause = final.error or next(
|
|
121
|
+
(t.last_error for t in final.traces if t.last_error), None
|
|
122
|
+
)
|
|
121
123
|
history.extend(final.errors)
|
|
122
124
|
for trace in final.traces:
|
|
123
125
|
history.extend(trace.errors)
|