verifiers 0.2.2.dev48__py3-none-any.whl → 0.2.2.dev49__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/agent.py +12 -12
- verifiers/v1/cli/dashboard/eval.py +13 -16
- verifiers/v1/cli/debug.py +2 -2
- verifiers/v1/cli/eval/runner.py +3 -3
- verifiers/v1/cli/replay.py +1 -1
- verifiers/v1/cli/validate.py +12 -0
- verifiers/v1/env.py +1 -1
- verifiers/v1/envs/agentic_judge/env.py +1 -1
- verifiers/v1/envs/user_sim/env.py +2 -2
- verifiers/v1/errors.py +1 -1
- verifiers/v1/gepa/adapter.py +2 -3
- verifiers/v1/legacy.py +6 -2
- verifiers/v1/push.py +5 -5
- verifiers/v1/retries.py +3 -1
- verifiers/v1/rollout.py +23 -20
- verifiers/v1/runtimes/subprocess.py +1 -1
- verifiers/v1/task.py +1 -1
- verifiers/v1/tasksets/harbor/taskset.py +4 -4
- verifiers/v1/trace.py +135 -238
- verifiers/v1/utils/compile.py +8 -8
- {verifiers-0.2.2.dev48.dist-info → verifiers-0.2.2.dev49.dist-info}/METADATA +1 -1
- {verifiers-0.2.2.dev48.dist-info → verifiers-0.2.2.dev49.dist-info}/RECORD +25 -25
- {verifiers-0.2.2.dev48.dist-info → verifiers-0.2.2.dev49.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev48.dist-info → verifiers-0.2.2.dev49.dist-info}/entry_points.txt +0 -0
- {verifiers-0.2.2.dev48.dist-info → verifiers-0.2.2.dev49.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/agent.py
CHANGED
|
@@ -45,7 +45,7 @@ from verifiers.v1.types import (
|
|
|
45
45
|
UserMessage,
|
|
46
46
|
)
|
|
47
47
|
from verifiers.v1.utils.compile import (
|
|
48
|
-
|
|
48
|
+
cap_remote_agent_timeout,
|
|
49
49
|
resolve_runtime_config,
|
|
50
50
|
validate_pairing,
|
|
51
51
|
)
|
|
@@ -390,7 +390,7 @@ class Agent:
|
|
|
390
390
|
attempt + 1,
|
|
391
391
|
retry.max_retries,
|
|
392
392
|
delay,
|
|
393
|
-
trace.
|
|
393
|
+
trace.last_error.type if trace.last_error else "?",
|
|
394
394
|
)
|
|
395
395
|
await asyncio.sleep(delay)
|
|
396
396
|
if history:
|
|
@@ -419,8 +419,8 @@ class Agent:
|
|
|
419
419
|
# close() never runs — free the run's servers and owned runtime first.
|
|
420
420
|
await run.abort()
|
|
421
421
|
raise
|
|
422
|
-
if trace.runtime is not None:
|
|
423
|
-
trace.runtime.borrowed = runtime is not None
|
|
422
|
+
if trace.agent.runtime is not None:
|
|
423
|
+
trace.agent.runtime.borrowed = runtime is not None
|
|
424
424
|
return trace
|
|
425
425
|
|
|
426
426
|
@asynccontextmanager
|
|
@@ -482,8 +482,8 @@ class Agent:
|
|
|
482
482
|
opened = await run.open()
|
|
483
483
|
if not opened:
|
|
484
484
|
trace = await run.close()
|
|
485
|
-
if trace.runtime is not None:
|
|
486
|
-
trace.runtime.borrowed = runtime is not None
|
|
485
|
+
if trace.agent.runtime is not None:
|
|
486
|
+
trace.agent.runtime.borrowed = runtime is not None
|
|
487
487
|
if not opened:
|
|
488
488
|
failure = run.failure
|
|
489
489
|
if failure is None: # `open()` returning False always captures one.
|
|
@@ -499,8 +499,8 @@ class Agent:
|
|
|
499
499
|
raise
|
|
500
500
|
finally:
|
|
501
501
|
trace = run.trace if run.closed else await interaction.close()
|
|
502
|
-
if trace.runtime is not None:
|
|
503
|
-
trace.runtime.borrowed = runtime is not None
|
|
502
|
+
if trace.agent.runtime is not None:
|
|
503
|
+
trace.agent.runtime.borrowed = runtime is not None
|
|
504
504
|
|
|
505
505
|
def _rollout_params(
|
|
506
506
|
self, task: Task, runtime: Runtime | None, shared_tools: dict
|
|
@@ -520,10 +520,10 @@ class Agent:
|
|
|
520
520
|
self.harness, type(task), runtime_config, shared_tools=shared_tools
|
|
521
521
|
)
|
|
522
522
|
# Timeout precedence: agent-level wins, else the task's, else no limit.
|
|
523
|
-
|
|
523
|
+
agent_timeout = (
|
|
524
524
|
self.timeout.rollout
|
|
525
525
|
if self.timeout.rollout is not None
|
|
526
|
-
else task.data.timeout.
|
|
526
|
+
else task.data.timeout.agent
|
|
527
527
|
)
|
|
528
528
|
return {
|
|
529
529
|
"agent_config": self.config,
|
|
@@ -535,8 +535,8 @@ class Agent:
|
|
|
535
535
|
if self.timeout.setup is not None
|
|
536
536
|
else task.data.timeout.setup
|
|
537
537
|
),
|
|
538
|
-
"
|
|
539
|
-
|
|
538
|
+
"agent_timeout": cap_remote_agent_timeout(
|
|
539
|
+
agent_timeout, runtime_config, task
|
|
540
540
|
),
|
|
541
541
|
"finalize_timeout": (
|
|
542
542
|
self.timeout.finalize
|
|
@@ -302,7 +302,7 @@ def Progress(
|
|
|
302
302
|
# run, a modeled user) are `trainable=False` and carry no rewards, so counting
|
|
303
303
|
# them dilutes every mean with structural zeros. An all-untrainable run (every
|
|
304
304
|
# role frozen) falls back to all traces rather than showing nothing.
|
|
305
|
-
scored = [t for t in done_traces if t.trainable] or done_traces
|
|
305
|
+
scored = [t for t in done_traces if t.agent.trainable] or done_traces
|
|
306
306
|
total = len(slots)
|
|
307
307
|
# Headline reward = mean over non-errored traces; when any errored, `format_mean` appends
|
|
308
308
|
# the global avg (errored count as 0) in parens. `err` is the share of episodes that
|
|
@@ -367,13 +367,13 @@ def _breakdown(scored: list[Trace], done: list[Trace]) -> Table | None:
|
|
|
367
367
|
# show); usage/time below still cover errored rollouts (their resources were spent regardless).
|
|
368
368
|
has_clean = any(not t.has_error for t in done)
|
|
369
369
|
score_rows = (("rewards", "rewards"), ("metrics", "metrics")) if has_clean else ()
|
|
370
|
-
by_agent: dict[str
|
|
370
|
+
by_agent: dict[str, list[Trace]] = {}
|
|
371
371
|
for trace in done:
|
|
372
|
-
by_agent.setdefault(trace.
|
|
372
|
+
by_agent.setdefault(trace.agent.name, []).append(trace)
|
|
373
373
|
for label, source in score_rows:
|
|
374
374
|
if len(by_agent) > 1:
|
|
375
375
|
segments = [
|
|
376
|
-
f"[dim]{name
|
|
376
|
+
f"[dim]{name}:[/dim] {means}"
|
|
377
377
|
for name, traces in by_agent.items()
|
|
378
378
|
if (means := _score_segments(traces, source)) is not None
|
|
379
379
|
]
|
|
@@ -505,7 +505,7 @@ def _stage(trace: Trace) -> str:
|
|
|
505
505
|
stage = "boot" # trace minted, first span not yet opened (an instant)
|
|
506
506
|
# A boot stuck on a first-use platform image build reads differently from a
|
|
507
507
|
# normal boot — it can sit there for ~10 minutes (prime runtime only).
|
|
508
|
-
if stage == "boot" and getattr(trace.runtime, "image_cached", None) is False:
|
|
508
|
+
if stage == "boot" and getattr(trace.agent.runtime, "image_cached", None) is False:
|
|
509
509
|
return "build"
|
|
510
510
|
return stage
|
|
511
511
|
|
|
@@ -564,14 +564,14 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
|
|
|
564
564
|
group_rows.append(("pending", [f"task {base}", *[""] * 7], "", ""))
|
|
565
565
|
continue
|
|
566
566
|
for t in slot.traces:
|
|
567
|
-
label = f"{base} agent={t.
|
|
567
|
+
label = f"{base} agent={t.agent.name}"
|
|
568
568
|
if slot.done: # fully scored — reward is final
|
|
569
569
|
state = "error" if t.has_error else "success"
|
|
570
570
|
# A trace that recorded nothing shows no reward: a judge or
|
|
571
571
|
# modeled-user seat's `reward=0.00` would read as a score.
|
|
572
572
|
result = (
|
|
573
|
-
t.
|
|
574
|
-
if t.has_error
|
|
573
|
+
t.last_error.type
|
|
574
|
+
if t.has_error and t.last_error
|
|
575
575
|
else (f"reward={t.reward:.2f}" if t.rewards else "")
|
|
576
576
|
)
|
|
577
577
|
if t.has_error:
|
|
@@ -582,7 +582,7 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
|
|
|
582
582
|
t.is_truncated
|
|
583
583
|
): # flag a clipped rollout next to its stop condition
|
|
584
584
|
stop = f"{stop} (truncated)".strip()
|
|
585
|
-
elif t.is_completed and (err := t.
|
|
585
|
+
elif t.is_completed and (err := t.last_error) is not None:
|
|
586
586
|
# An errored trace whose episode is still running its other
|
|
587
587
|
# traces (or `score()`) is already a failure — show it, don't
|
|
588
588
|
# let it sit as "scoring" until the whole episode lands.
|
|
@@ -592,12 +592,9 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
|
|
|
592
592
|
# The trace's own stamp, not the run-level runtime: a role's harness
|
|
593
593
|
# may resolve elsewhere (the judge env's sandboxed judge on a
|
|
594
594
|
# subprocess run).
|
|
595
|
-
if t.runtime is not None:
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
if t.runtime.id
|
|
599
|
-
else t.runtime.type
|
|
600
|
-
)
|
|
595
|
+
if t.agent.runtime is not None:
|
|
596
|
+
rt = t.agent.runtime
|
|
597
|
+
runtime = f"{rt.type}({rt.id})" if rt.id else rt.type
|
|
601
598
|
else:
|
|
602
599
|
runtime = runtime_type
|
|
603
600
|
turns = t.num_turns
|
|
@@ -635,7 +632,7 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
|
|
|
635
632
|
f"{nbranches} branch{'es' * (nbranches != 1)}",
|
|
636
633
|
tokens,
|
|
637
634
|
f"{format_cost_usd(cost)}" if cost is not None else "",
|
|
638
|
-
stop, # stop condition (agent_completed / max_turns /
|
|
635
|
+
stop, # stop condition (agent_completed / max_turns / error), once done
|
|
639
636
|
]
|
|
640
637
|
# No start time yet (queued, not generating) → blank, not `now - 0` (~56 years).
|
|
641
638
|
elapsed = format_time(end - start) if start else ""
|
verifiers/v1/cli/debug.py
CHANGED
|
@@ -101,9 +101,9 @@ def error_info(
|
|
|
101
101
|
|
|
102
102
|
|
|
103
103
|
def capture_trace_error(trace: Trace, error: BaseException) -> None:
|
|
104
|
-
# CancelledError is a BaseException; Trace.
|
|
104
|
+
# CancelledError is a BaseException; Trace.record_error accepts Exception.
|
|
105
105
|
if isinstance(error, Exception):
|
|
106
|
-
trace.
|
|
106
|
+
trace.record_error(error)
|
|
107
107
|
return
|
|
108
108
|
trace.errors.append(
|
|
109
109
|
Error(
|
verifiers/v1/cli/eval/runner.py
CHANGED
|
@@ -68,7 +68,7 @@ async def run_eval(env: Env, config: EvalConfig) -> list[Episode]:
|
|
|
68
68
|
|
|
69
69
|
async def on_complete(episode: Episode) -> None:
|
|
70
70
|
for trace in episode.traces:
|
|
71
|
-
trace.
|
|
71
|
+
trace.record_run(EvalRunInfo(id=config.uuid))
|
|
72
72
|
await append_episode(out, episode, write_lock)
|
|
73
73
|
|
|
74
74
|
# Serving resources (shared tool servers, interception) come up once for the
|
|
@@ -240,7 +240,7 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
|
|
|
240
240
|
)
|
|
241
241
|
records = []
|
|
242
242
|
for trace in traces:
|
|
243
|
-
trace.
|
|
243
|
+
trace.record_run(EvalRunInfo(id=config.uuid))
|
|
244
244
|
await append_trace(out, trace, write_lock, env=config.env_id)
|
|
245
245
|
records.append(Episode.of(trace))
|
|
246
246
|
return records
|
|
@@ -254,7 +254,7 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
|
|
|
254
254
|
**payload,
|
|
255
255
|
)
|
|
256
256
|
for trace in episode.traces:
|
|
257
|
-
trace.
|
|
257
|
+
trace.record_run(EvalRunInfo(id=config.uuid))
|
|
258
258
|
await append_episode(out, episode, write_lock)
|
|
259
259
|
return [episode]
|
|
260
260
|
|
verifiers/v1/cli/replay.py
CHANGED
|
@@ -165,7 +165,7 @@ async def run_replay(config: ReplayConfig, source: Path, out: Path) -> list[Trac
|
|
|
165
165
|
st.state, st.detail = "scored", f"reward {trace.reward:.3f}"
|
|
166
166
|
except Exception as exc:
|
|
167
167
|
st.state, st.detail = "error", type(exc).__name__
|
|
168
|
-
trace.
|
|
168
|
+
trace.record_error(exc)
|
|
169
169
|
if not config.rich:
|
|
170
170
|
logger.warning(
|
|
171
171
|
"replay: scoring failed for task %s",
|
verifiers/v1/cli/validate.py
CHANGED
|
@@ -89,6 +89,12 @@ async def _run_gold(task: Task, config: ValidateConfig) -> ResultRow:
|
|
|
89
89
|
trace = Trace(
|
|
90
90
|
task=TraceTask(type=type(task).__name__, data=task.data),
|
|
91
91
|
state=state_cls(type(task))(),
|
|
92
|
+
# No agent runs here — the info only records the runtime policy.
|
|
93
|
+
agent=vf.AgentInfo(
|
|
94
|
+
config=vf.AgentConfig(runtime=config.runtime),
|
|
95
|
+
name="validate",
|
|
96
|
+
trainable=False,
|
|
97
|
+
),
|
|
92
98
|
)
|
|
93
99
|
await runtime.start()
|
|
94
100
|
await asyncio.wait_for(
|
|
@@ -124,6 +130,12 @@ async def _run_setup(task: Task, config: ValidateConfig) -> ResultRow:
|
|
|
124
130
|
trace = Trace(
|
|
125
131
|
task=TraceTask(type=type(task).__name__, data=task.data),
|
|
126
132
|
state=state_cls(type(task))(),
|
|
133
|
+
# No agent runs here — the info only records the runtime policy.
|
|
134
|
+
agent=vf.AgentInfo(
|
|
135
|
+
config=vf.AgentConfig(runtime=config.runtime),
|
|
136
|
+
name="validate",
|
|
137
|
+
trainable=False,
|
|
138
|
+
),
|
|
127
139
|
)
|
|
128
140
|
await runtime.start()
|
|
129
141
|
await asyncio.wait_for(
|
verifiers/v1/env.py
CHANGED
|
@@ -163,7 +163,7 @@ class Env(ABC, Generic[ConfigT]):
|
|
|
163
163
|
"""Cross-agent judgement — THE programmable judgement surface: plain
|
|
164
164
|
imperative Python over the finished episode (per-trace judgement already
|
|
165
165
|
ran on each trace's own task). `episode.traces` is the flat episode in
|
|
166
|
-
completion order, each trace's `
|
|
166
|
+
completion order, each trace's `agent.name` stamp naming its agent; attach
|
|
167
167
|
signals via `record_reward`/`record_metric`, in program order. A raise
|
|
168
168
|
fails the episode (the retryable unit) — validate strictly, never
|
|
169
169
|
record a guess."""
|
|
@@ -304,7 +304,7 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
|
|
|
304
304
|
await agents.judge.run(judge_task, runtime=box)
|
|
305
305
|
|
|
306
306
|
async def finalize(self, task: vf.Task, episode: vf.Episode) -> None:
|
|
307
|
-
by_agent = {t.
|
|
307
|
+
by_agent = {t.agent.name: t for t in episode.traces}
|
|
308
308
|
solution, verdict = by_agent["solver"], by_agent["judge"]
|
|
309
309
|
data = verdict.info.get("verdict")
|
|
310
310
|
if not isinstance(data, dict) or not isinstance(data.get("verdicts"), list):
|
|
@@ -89,7 +89,7 @@ class UserSimEnv(vf.Env[UserSimEnvConfig]):
|
|
|
89
89
|
async def finalize(self, task, episode):
|
|
90
90
|
"""One conversation-shape fact about the user's side, recorded on the
|
|
91
91
|
assistant's trace; judgement stays on the task's rewards."""
|
|
92
|
-
(user,) = (t for t in episode.traces if t.
|
|
92
|
+
(user,) = (t for t in episode.traces if t.agent.name == "user")
|
|
93
93
|
for trace in episode.traces:
|
|
94
|
-
if trace.
|
|
94
|
+
if trace.agent.name == "assistant":
|
|
95
95
|
trace.record_metric("user_turns", float(user.num_turns))
|
verifiers/v1/errors.py
CHANGED
|
@@ -4,7 +4,7 @@ Four mechanisms, each in one place:
|
|
|
4
4
|
|
|
5
5
|
1. Vocabulary (this module): `RolloutError` and the flat boundary types below. Each names the
|
|
6
6
|
boundary a failure crossed — provider, harness, toolset, sandbox, task, or
|
|
7
|
-
interception — so a recorded `trace.
|
|
7
|
+
interception — so a recorded `trace.last_error.type` says where the rollout broke.
|
|
8
8
|
2. Classification (`boundary`): the one helper that runs a framework→code boundary and attributes
|
|
9
9
|
any escaping error to that boundary's type. Extension code (task hooks, harness subclasses)
|
|
10
10
|
raises plain Python errors — it never constructs a `vf` error type; `boundary` classifies them.
|
verifiers/v1/gepa/adapter.py
CHANGED
|
@@ -104,10 +104,9 @@ class GEPAAdapter:
|
|
|
104
104
|
"completion": trace.last_reply,
|
|
105
105
|
"reward": trace.reward,
|
|
106
106
|
}
|
|
107
|
-
|
|
108
|
-
record["agent"] = trace.agent_name
|
|
107
|
+
record["agent"] = trace.agent.name
|
|
109
108
|
if trace.has_error:
|
|
110
|
-
record["error"] = str(trace.
|
|
109
|
+
record["error"] = str(trace.last_error)
|
|
111
110
|
if trace.stop_condition:
|
|
112
111
|
record["stop_condition"] = trace.stop_condition
|
|
113
112
|
for column in self.reflection_columns:
|
verifiers/v1/legacy.py
CHANGED
|
@@ -24,6 +24,7 @@ from pydantic import ValidationError
|
|
|
24
24
|
|
|
25
25
|
from verifiers.v1 import graph
|
|
26
26
|
from verifiers.v1.clients.config import ClientConfig, TrainClientConfig
|
|
27
|
+
from verifiers.v1.configs.agent import AgentConfig
|
|
27
28
|
from verifiers.v1.episode import Episode
|
|
28
29
|
from verifiers.v1.serve.server import EnvServer
|
|
29
30
|
from verifiers.v1.serve.types import (
|
|
@@ -34,6 +35,7 @@ from verifiers.v1.serve.types import (
|
|
|
34
35
|
)
|
|
35
36
|
from verifiers.v1.task import WireTaskData
|
|
36
37
|
from verifiers.v1.trace import (
|
|
38
|
+
AgentInfo,
|
|
37
39
|
Error,
|
|
38
40
|
GenerationSpan,
|
|
39
41
|
ModelCall,
|
|
@@ -227,7 +229,6 @@ def _timing(raw: Any) -> Timing:
|
|
|
227
229
|
_V0_TO_V1_TRUNCATION_STOP = {
|
|
228
230
|
"max_turns_reached": "max_turns",
|
|
229
231
|
"prompt_too_long": "context_length",
|
|
230
|
-
"timeout_reached": "harness_timeout",
|
|
231
232
|
"max_total_completion_tokens_reached": "max_output_tokens",
|
|
232
233
|
}
|
|
233
234
|
|
|
@@ -270,7 +271,10 @@ def rollout_output_to_trace(out: dict, task_idx: int) -> Trace:
|
|
|
270
271
|
type="Task",
|
|
271
272
|
data=_to_wire_task(task_idx, out.get("prompt"), out.get("answer")),
|
|
272
273
|
),
|
|
273
|
-
|
|
274
|
+
# v0 rollouts carry no agent config — record the default, like the
|
|
275
|
+
# base task type above.
|
|
276
|
+
agent=AgentInfo(config=AgentConfig()),
|
|
277
|
+
tools=_to_v1_tools(out.get("tool_defs")) or [],
|
|
274
278
|
rewards={"reward": Reward(score=float(out.get("reward") or 0.0))},
|
|
275
279
|
metrics={k: float(v) for k, v in (out.get("metrics") or {}).items()},
|
|
276
280
|
info=dict(out.get("info") or {}),
|
verifiers/v1/push.py
CHANGED
|
@@ -58,8 +58,8 @@ def trace_to_sample(
|
|
|
58
58
|
"example_id": trace.task.data.idx,
|
|
59
59
|
"rollout_number": rollout_number,
|
|
60
60
|
"episode_id": episode_id,
|
|
61
|
-
"agent": trace.
|
|
62
|
-
"trainable": trace.trainable,
|
|
61
|
+
"agent": trace.agent.name,
|
|
62
|
+
"trainable": trace.agent.trainable,
|
|
63
63
|
"task": task,
|
|
64
64
|
"prompt": [],
|
|
65
65
|
"completion": dump(branches[-1].messages) if branches else [],
|
|
@@ -73,8 +73,8 @@ def trace_to_sample(
|
|
|
73
73
|
"is_completed": trace.is_completed,
|
|
74
74
|
"is_truncated": trace.is_truncated,
|
|
75
75
|
"metrics": trace.metrics,
|
|
76
|
-
"error": trace.
|
|
77
|
-
if trace.
|
|
76
|
+
"error": trace.last_error.model_dump(mode="json", exclude_none=True)
|
|
77
|
+
if trace.last_error
|
|
78
78
|
else None,
|
|
79
79
|
"stop_condition": trace.stop_condition,
|
|
80
80
|
"trajectory": [
|
|
@@ -124,7 +124,7 @@ def _run_metrics(episodes: list[Episode], traces: list[Trace]) -> dict[str, Any]
|
|
|
124
124
|
back to all traces when none are trainable (same rule as the dashboard).
|
|
125
125
|
`avg_error` is the share of EPISODES that aren't ok: a hook failure counts
|
|
126
126
|
even when its traces are clean or it left none."""
|
|
127
|
-
scored = [t for t in traces if t.trainable] or traces
|
|
127
|
+
scored = [t for t in traces if t.agent.trainable] or traces
|
|
128
128
|
sums: dict[str, float] = {}
|
|
129
129
|
counts: dict[str, int] = {}
|
|
130
130
|
for trace in scored:
|
verifiers/v1/retries.py
CHANGED
|
@@ -117,7 +117,9 @@ async def run_episode_with_retry(
|
|
|
117
117
|
final = await run()
|
|
118
118
|
if attempt == retry.max_retries or not episode_should_retry(final, retry):
|
|
119
119
|
break
|
|
120
|
-
cause = final.error or next(
|
|
120
|
+
cause = final.error or next(
|
|
121
|
+
(t.last_error for t in final.traces if t.last_error), None
|
|
122
|
+
)
|
|
121
123
|
history.extend(final.errors)
|
|
122
124
|
for trace in final.traces:
|
|
123
125
|
history.extend(trace.errors)
|
verifiers/v1/rollout.py
CHANGED
|
@@ -24,7 +24,6 @@ import time
|
|
|
24
24
|
from collections.abc import AsyncIterator, Callable
|
|
25
25
|
from contextlib import AsyncExitStack, asynccontextmanager
|
|
26
26
|
|
|
27
|
-
from verifiers import __version__
|
|
28
27
|
from verifiers.v1.clients import ModelContext
|
|
29
28
|
from verifiers.v1.configs.agent import AgentConfig
|
|
30
29
|
from verifiers.v1.decorators import discover_decorated, invoke
|
|
@@ -52,9 +51,8 @@ from verifiers.v1.runtimes import (
|
|
|
52
51
|
from verifiers.v1.session import RolloutLimits, RolloutSession
|
|
53
52
|
from verifiers.v1.state import state_cls
|
|
54
53
|
from verifiers.v1.task import Task, TaskData
|
|
55
|
-
from verifiers.v1.trace import AgentInfo, Trace, TraceTask
|
|
54
|
+
from verifiers.v1.trace import AgentInfo, Trace, TraceTask
|
|
56
55
|
from verifiers.v1.types import Messages
|
|
57
|
-
from verifiers.v1.utils.version import verifiers_commit
|
|
58
56
|
|
|
59
57
|
logger = logging.getLogger(__name__)
|
|
60
58
|
|
|
@@ -118,7 +116,7 @@ class RolloutRun:
|
|
|
118
116
|
wire_data: TaskData | None = None,
|
|
119
117
|
has_user: bool = False,
|
|
120
118
|
setup_timeout: float | None = None,
|
|
121
|
-
|
|
119
|
+
agent_timeout: float | None = None,
|
|
122
120
|
finalize_timeout: float | None = None,
|
|
123
121
|
scoring_timeout: float | None = None,
|
|
124
122
|
limits: RolloutLimits | None = None,
|
|
@@ -133,7 +131,8 @@ class RolloutRun:
|
|
|
133
131
|
self.runtime_config = runtime_config
|
|
134
132
|
self._has_user = has_user
|
|
135
133
|
self._setup_timeout = setup_timeout
|
|
136
|
-
self.
|
|
134
|
+
self._agent_timeout = agent_timeout
|
|
135
|
+
self._agent_time_remaining = agent_timeout
|
|
137
136
|
self._finalize_timeout = finalize_timeout
|
|
138
137
|
self._scoring_timeout = scoring_timeout
|
|
139
138
|
self._shared_tools = shared_tools or {}
|
|
@@ -146,7 +145,6 @@ class RolloutRun:
|
|
|
146
145
|
data=task.data if wire_data is None else wire_data,
|
|
147
146
|
),
|
|
148
147
|
state=state_cls(type(task))(),
|
|
149
|
-
verifiers=VersionInfo(version=__version__, commit=verifiers_commit()),
|
|
150
148
|
# The seat's resolved config, role overrides included — the agent
|
|
151
149
|
# this trace can be reproduced with.
|
|
152
150
|
agent=AgentInfo(config=agent_config),
|
|
@@ -166,7 +164,7 @@ class RolloutRun:
|
|
|
166
164
|
self.deadline_at: float | None = None
|
|
167
165
|
"""The active harness segment's absolute deadline (event-loop clock), or
|
|
168
166
|
None between segments / when unbounded. An interaction spends one cumulative
|
|
169
|
-
`
|
|
167
|
+
`agent_timeout` budget only while its own segments run, so time awaiting
|
|
170
168
|
the caller (including another interleaved agent) cannot starve it."""
|
|
171
169
|
|
|
172
170
|
@property
|
|
@@ -201,7 +199,7 @@ class RolloutRun:
|
|
|
201
199
|
logger.exception("unexpected error in rollout %s", self.trace.id)
|
|
202
200
|
self._failed = True
|
|
203
201
|
self._failure = error
|
|
204
|
-
self.trace.
|
|
202
|
+
self.trace.record_error(error)
|
|
205
203
|
|
|
206
204
|
async def open(self) -> bool:
|
|
207
205
|
"""Boot the rollout's world up to the point where segments can run: start
|
|
@@ -308,7 +306,8 @@ class RolloutRun:
|
|
|
308
306
|
for an exchange the user opens, this is also the first segment, on an
|
|
309
307
|
empty conversation); without, it launches on the task's own prompt.
|
|
310
308
|
Returns whether the exchange can continue — a refused turn (limit, @stop),
|
|
311
|
-
a timeout
|
|
309
|
+
a failure (an expired agent timeout included), or a segment that made no
|
|
310
|
+
progress all end it."""
|
|
312
311
|
if not self._opened or self._closed or not self.ok:
|
|
313
312
|
return False
|
|
314
313
|
trace = self.trace
|
|
@@ -317,11 +316,10 @@ class RolloutRun:
|
|
|
317
316
|
segment_start = loop.time()
|
|
318
317
|
self.deadline_at = (
|
|
319
318
|
None
|
|
320
|
-
if self.
|
|
321
|
-
else segment_start + max(0.0, self.
|
|
319
|
+
if self._agent_time_remaining is None
|
|
320
|
+
else segment_start + max(0.0, self._agent_time_remaining)
|
|
322
321
|
)
|
|
323
322
|
# Prefer an intercepted model/tool error to the harness exit it caused.
|
|
324
|
-
# A timeout still scores the partial trajectory.
|
|
325
323
|
try:
|
|
326
324
|
async with asyncio.timeout_at(self.deadline_at):
|
|
327
325
|
await self.harness.run(
|
|
@@ -335,11 +333,16 @@ class RolloutRun:
|
|
|
335
333
|
messages,
|
|
336
334
|
)
|
|
337
335
|
except TimeoutError as e:
|
|
338
|
-
#
|
|
339
|
-
#
|
|
340
|
-
#
|
|
336
|
+
# An expired rollout deadline is the agent breaking its time budget —
|
|
337
|
+
# an agent failure, never a clean stop. A TimeoutError from the
|
|
338
|
+
# harness's own I/O with no expired deadline stays the raw failure.
|
|
341
339
|
if self.deadline_at is not None and (loop.time() >= self.deadline_at):
|
|
342
|
-
|
|
340
|
+
self.fail(
|
|
341
|
+
HarnessError(
|
|
342
|
+
f"agent timeout: rollout exceeded its "
|
|
343
|
+
f"{self._agent_timeout:g}s budget"
|
|
344
|
+
)
|
|
345
|
+
)
|
|
343
346
|
else:
|
|
344
347
|
self.fail(e)
|
|
345
348
|
return False
|
|
@@ -352,9 +355,9 @@ class RolloutRun:
|
|
|
352
355
|
self.fail(e)
|
|
353
356
|
return False
|
|
354
357
|
finally:
|
|
355
|
-
if self.
|
|
356
|
-
self.
|
|
357
|
-
0.0, self.
|
|
358
|
+
if self._agent_time_remaining is not None:
|
|
359
|
+
self._agent_time_remaining = max(
|
|
360
|
+
0.0, self._agent_time_remaining - (loop.time() - segment_start)
|
|
358
361
|
)
|
|
359
362
|
self.deadline_at = None
|
|
360
363
|
if self._session.error is not None:
|
|
@@ -457,6 +460,6 @@ class RolloutRun:
|
|
|
457
460
|
self.task.data.idx,
|
|
458
461
|
trace.reward,
|
|
459
462
|
trace.num_turns,
|
|
460
|
-
trace.
|
|
463
|
+
trace.last_error.type if trace.last_error else trace.stop_condition,
|
|
461
464
|
)
|
|
462
465
|
return trace
|
|
@@ -62,7 +62,7 @@ class SubprocessRuntime(Runtime):
|
|
|
62
62
|
stdout, stderr = await proc.communicate()
|
|
63
63
|
finally:
|
|
64
64
|
# If the await didn't finish, the caller cancelled it (e.g. the rollout's
|
|
65
|
-
# scoring_timeout /
|
|
65
|
+
# scoring_timeout / agent_timeout fired): communicate() leaves the process
|
|
66
66
|
# running, so SIGKILL its whole group (start_new_session => pgid == pid) — otherwise
|
|
67
67
|
# a hung child (a wedged uv/sympy verify) outlives the rollout and leaks CPU. A
|
|
68
68
|
# no-op once it has exited on its own.
|
verifiers/v1/task.py
CHANGED
|
@@ -260,9 +260,9 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
|
|
|
260
260
|
if not authors and meta.get("author_name"):
|
|
261
261
|
authors = [Author(name=meta["author_name"], email=meta.get("author_email"))]
|
|
262
262
|
if harbor_config.ignore_timeouts:
|
|
263
|
-
|
|
263
|
+
agent_timeout = scoring_timeout = None
|
|
264
264
|
else:
|
|
265
|
-
|
|
265
|
+
agent_timeout = (
|
|
266
266
|
parsed.agent.timeout_sec
|
|
267
267
|
if "timeout_sec" in parsed.agent.model_fields_set
|
|
268
268
|
else None
|
|
@@ -290,8 +290,8 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
|
|
|
290
290
|
else list(network.allowed_hosts)
|
|
291
291
|
),
|
|
292
292
|
timeout=TaskTimeout(
|
|
293
|
-
|
|
294
|
-
if
|
|
293
|
+
agent=agent_timeout * harbor_config.timeout_multiplier
|
|
294
|
+
if agent_timeout is not None
|
|
295
295
|
else None,
|
|
296
296
|
scoring=scoring_timeout * harbor_config.timeout_multiplier
|
|
297
297
|
if scoring_timeout is not None
|