verifiers 0.2.2.dev48__py3-none-any.whl → 0.2.2.dev49__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
verifiers/v1/agent.py CHANGED
@@ -45,7 +45,7 @@ from verifiers.v1.types import (
45
45
  UserMessage,
46
46
  )
47
47
  from verifiers.v1.utils.compile import (
48
- cap_remote_harness_timeout,
48
+ cap_remote_agent_timeout,
49
49
  resolve_runtime_config,
50
50
  validate_pairing,
51
51
  )
@@ -390,7 +390,7 @@ class Agent:
390
390
  attempt + 1,
391
391
  retry.max_retries,
392
392
  delay,
393
- trace.error.type if trace.error else "?",
393
+ trace.last_error.type if trace.last_error else "?",
394
394
  )
395
395
  await asyncio.sleep(delay)
396
396
  if history:
@@ -419,8 +419,8 @@ class Agent:
419
419
  # close() never runs — free the run's servers and owned runtime first.
420
420
  await run.abort()
421
421
  raise
422
- if trace.runtime is not None:
423
- trace.runtime.borrowed = runtime is not None
422
+ if trace.agent.runtime is not None:
423
+ trace.agent.runtime.borrowed = runtime is not None
424
424
  return trace
425
425
 
426
426
  @asynccontextmanager
@@ -482,8 +482,8 @@ class Agent:
482
482
  opened = await run.open()
483
483
  if not opened:
484
484
  trace = await run.close()
485
- if trace.runtime is not None:
486
- trace.runtime.borrowed = runtime is not None
485
+ if trace.agent.runtime is not None:
486
+ trace.agent.runtime.borrowed = runtime is not None
487
487
  if not opened:
488
488
  failure = run.failure
489
489
  if failure is None: # `open()` returning False always captures one.
@@ -499,8 +499,8 @@ class Agent:
499
499
  raise
500
500
  finally:
501
501
  trace = run.trace if run.closed else await interaction.close()
502
- if trace.runtime is not None:
503
- trace.runtime.borrowed = runtime is not None
502
+ if trace.agent.runtime is not None:
503
+ trace.agent.runtime.borrowed = runtime is not None
504
504
 
505
505
  def _rollout_params(
506
506
  self, task: Task, runtime: Runtime | None, shared_tools: dict
@@ -520,10 +520,10 @@ class Agent:
520
520
  self.harness, type(task), runtime_config, shared_tools=shared_tools
521
521
  )
522
522
  # Timeout precedence: agent-level wins, else the task's, else no limit.
523
- harness_timeout = (
523
+ agent_timeout = (
524
524
  self.timeout.rollout
525
525
  if self.timeout.rollout is not None
526
- else task.data.timeout.harness
526
+ else task.data.timeout.agent
527
527
  )
528
528
  return {
529
529
  "agent_config": self.config,
@@ -535,8 +535,8 @@ class Agent:
535
535
  if self.timeout.setup is not None
536
536
  else task.data.timeout.setup
537
537
  ),
538
- "harness_timeout": cap_remote_harness_timeout(
539
- harness_timeout, runtime_config, task
538
+ "agent_timeout": cap_remote_agent_timeout(
539
+ agent_timeout, runtime_config, task
540
540
  ),
541
541
  "finalize_timeout": (
542
542
  self.timeout.finalize
@@ -302,7 +302,7 @@ def Progress(
302
302
  # run, a modeled user) are `trainable=False` and carry no rewards, so counting
303
303
  # them dilutes every mean with structural zeros. An all-untrainable run (every
304
304
  # role frozen) falls back to all traces rather than showing nothing.
305
- scored = [t for t in done_traces if t.trainable] or done_traces
305
+ scored = [t for t in done_traces if t.agent.trainable] or done_traces
306
306
  total = len(slots)
307
307
  # Headline reward = mean over non-errored traces; when any errored, `format_mean` appends
308
308
  # the global avg (errored count as 0) in parens. `err` is the share of episodes that
@@ -367,13 +367,13 @@ def _breakdown(scored: list[Trace], done: list[Trace]) -> Table | None:
367
367
  # show); usage/time below still cover errored rollouts (their resources were spent regardless).
368
368
  has_clean = any(not t.has_error for t in done)
369
369
  score_rows = (("rewards", "rewards"), ("metrics", "metrics")) if has_clean else ()
370
- by_agent: dict[str | None, list[Trace]] = {}
370
+ by_agent: dict[str, list[Trace]] = {}
371
371
  for trace in done:
372
- by_agent.setdefault(trace.agent_name, []).append(trace)
372
+ by_agent.setdefault(trace.agent.name, []).append(trace)
373
373
  for label, source in score_rows:
374
374
  if len(by_agent) > 1:
375
375
  segments = [
376
- f"[dim]{name or '—'}:[/dim] {means}"
376
+ f"[dim]{name}:[/dim] {means}"
377
377
  for name, traces in by_agent.items()
378
378
  if (means := _score_segments(traces, source)) is not None
379
379
  ]
@@ -505,7 +505,7 @@ def _stage(trace: Trace) -> str:
505
505
  stage = "boot" # trace minted, first span not yet opened (an instant)
506
506
  # A boot stuck on a first-use platform image build reads differently from a
507
507
  # normal boot — it can sit there for ~10 minutes (prime runtime only).
508
- if stage == "boot" and getattr(trace.runtime, "image_cached", None) is False:
508
+ if stage == "boot" and getattr(trace.agent.runtime, "image_cached", None) is False:
509
509
  return "build"
510
510
  return stage
511
511
 
@@ -564,14 +564,14 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
564
564
  group_rows.append(("pending", [f"task {base}", *[""] * 7], "", ""))
565
565
  continue
566
566
  for t in slot.traces:
567
- label = f"{base} agent={t.agent_name}" if t.agent_name else base
567
+ label = f"{base} agent={t.agent.name}"
568
568
  if slot.done: # fully scored — reward is final
569
569
  state = "error" if t.has_error else "success"
570
570
  # A trace that recorded nothing shows no reward: a judge or
571
571
  # modeled-user seat's `reward=0.00` would read as a score.
572
572
  result = (
573
- t.error.type
574
- if t.has_error
573
+ t.last_error.type
574
+ if t.has_error and t.last_error
575
575
  else (f"reward={t.reward:.2f}" if t.rewards else "")
576
576
  )
577
577
  if t.has_error:
@@ -582,7 +582,7 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
582
582
  t.is_truncated
583
583
  ): # flag a clipped rollout next to its stop condition
584
584
  stop = f"{stop} (truncated)".strip()
585
- elif t.is_completed and (err := t.error) is not None:
585
+ elif t.is_completed and (err := t.last_error) is not None:
586
586
  # An errored trace whose episode is still running its other
587
587
  # traces (or `score()`) is already a failure — show it, don't
588
588
  # let it sit as "scoring" until the whole episode lands.
@@ -592,12 +592,9 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
592
592
  # The trace's own stamp, not the run-level runtime: a role's harness
593
593
  # may resolve elsewhere (the judge env's sandboxed judge on a
594
594
  # subprocess run).
595
- if t.runtime is not None:
596
- runtime = (
597
- f"{t.runtime.type}({t.runtime.id})"
598
- if t.runtime.id
599
- else t.runtime.type
600
- )
595
+ if t.agent.runtime is not None:
596
+ rt = t.agent.runtime
597
+ runtime = f"{rt.type}({rt.id})" if rt.id else rt.type
601
598
  else:
602
599
  runtime = runtime_type
603
600
  turns = t.num_turns
@@ -635,7 +632,7 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
635
632
  f"{nbranches} branch{'es' * (nbranches != 1)}",
636
633
  tokens,
637
634
  f"{format_cost_usd(cost)}" if cost is not None else "",
638
- stop, # stop condition (agent_completed / max_turns / harness_timeout), once done
635
+ stop, # stop condition (agent_completed / max_turns / error), once done
639
636
  ]
640
637
  # No start time yet (queued, not generating) → blank, not `now - 0` (~56 years).
641
638
  elapsed = format_time(end - start) if start else ""
verifiers/v1/cli/debug.py CHANGED
@@ -101,9 +101,9 @@ def error_info(
101
101
 
102
102
 
103
103
  def capture_trace_error(trace: Trace, error: BaseException) -> None:
104
- # CancelledError is a BaseException; Trace.capture_error accepts Exception.
104
+ # CancelledError is a BaseException; Trace.record_error accepts Exception.
105
105
  if isinstance(error, Exception):
106
- trace.capture_error(error)
106
+ trace.record_error(error)
107
107
  return
108
108
  trace.errors.append(
109
109
  Error(
@@ -68,7 +68,7 @@ async def run_eval(env: Env, config: EvalConfig) -> list[Episode]:
68
68
 
69
69
  async def on_complete(episode: Episode) -> None:
70
70
  for trace in episode.traces:
71
- trace.stamp(EvalRunInfo(id=config.uuid))
71
+ trace.record_run(EvalRunInfo(id=config.uuid))
72
72
  await append_episode(out, episode, write_lock)
73
73
 
74
74
  # Serving resources (shared tool servers, interception) come up once for the
@@ -240,7 +240,7 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
240
240
  )
241
241
  records = []
242
242
  for trace in traces:
243
- trace.stamp(EvalRunInfo(id=config.uuid))
243
+ trace.record_run(EvalRunInfo(id=config.uuid))
244
244
  await append_trace(out, trace, write_lock, env=config.env_id)
245
245
  records.append(Episode.of(trace))
246
246
  return records
@@ -254,7 +254,7 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
254
254
  **payload,
255
255
  )
256
256
  for trace in episode.traces:
257
- trace.stamp(EvalRunInfo(id=config.uuid))
257
+ trace.record_run(EvalRunInfo(id=config.uuid))
258
258
  await append_episode(out, episode, write_lock)
259
259
  return [episode]
260
260
 
@@ -165,7 +165,7 @@ async def run_replay(config: ReplayConfig, source: Path, out: Path) -> list[Trac
165
165
  st.state, st.detail = "scored", f"reward {trace.reward:.3f}"
166
166
  except Exception as exc:
167
167
  st.state, st.detail = "error", type(exc).__name__
168
- trace.capture_error(exc)
168
+ trace.record_error(exc)
169
169
  if not config.rich:
170
170
  logger.warning(
171
171
  "replay: scoring failed for task %s",
@@ -89,6 +89,12 @@ async def _run_gold(task: Task, config: ValidateConfig) -> ResultRow:
89
89
  trace = Trace(
90
90
  task=TraceTask(type=type(task).__name__, data=task.data),
91
91
  state=state_cls(type(task))(),
92
+ # No agent runs here — the info only records the runtime policy.
93
+ agent=vf.AgentInfo(
94
+ config=vf.AgentConfig(runtime=config.runtime),
95
+ name="validate",
96
+ trainable=False,
97
+ ),
92
98
  )
93
99
  await runtime.start()
94
100
  await asyncio.wait_for(
@@ -124,6 +130,12 @@ async def _run_setup(task: Task, config: ValidateConfig) -> ResultRow:
124
130
  trace = Trace(
125
131
  task=TraceTask(type=type(task).__name__, data=task.data),
126
132
  state=state_cls(type(task))(),
133
+ # No agent runs here — the info only records the runtime policy.
134
+ agent=vf.AgentInfo(
135
+ config=vf.AgentConfig(runtime=config.runtime),
136
+ name="validate",
137
+ trainable=False,
138
+ ),
127
139
  )
128
140
  await runtime.start()
129
141
  await asyncio.wait_for(
verifiers/v1/env.py CHANGED
@@ -163,7 +163,7 @@ class Env(ABC, Generic[ConfigT]):
163
163
  """Cross-agent judgement — THE programmable judgement surface: plain
164
164
  imperative Python over the finished episode (per-trace judgement already
165
165
  ran on each trace's own task). `episode.traces` is the flat episode in
166
- completion order, each trace's `agent_name` stamp naming its agent; attach
166
+ completion order, each trace's `agent.name` stamp naming its agent; attach
167
167
  signals via `record_reward`/`record_metric`, in program order. A raise
168
168
  fails the episode (the retryable unit) — validate strictly, never
169
169
  record a guess."""
@@ -304,7 +304,7 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
304
304
  await agents.judge.run(judge_task, runtime=box)
305
305
 
306
306
  async def finalize(self, task: vf.Task, episode: vf.Episode) -> None:
307
- by_agent = {t.agent_name: t for t in episode.traces}
307
+ by_agent = {t.agent.name: t for t in episode.traces}
308
308
  solution, verdict = by_agent["solver"], by_agent["judge"]
309
309
  data = verdict.info.get("verdict")
310
310
  if not isinstance(data, dict) or not isinstance(data.get("verdicts"), list):
@@ -89,7 +89,7 @@ class UserSimEnv(vf.Env[UserSimEnvConfig]):
89
89
  async def finalize(self, task, episode):
90
90
  """One conversation-shape fact about the user's side, recorded on the
91
91
  assistant's trace; judgement stays on the task's rewards."""
92
- (user,) = (t for t in episode.traces if t.agent_name == "user")
92
+ (user,) = (t for t in episode.traces if t.agent.name == "user")
93
93
  for trace in episode.traces:
94
- if trace.agent_name == "assistant":
94
+ if trace.agent.name == "assistant":
95
95
  trace.record_metric("user_turns", float(user.num_turns))
verifiers/v1/errors.py CHANGED
@@ -4,7 +4,7 @@ Four mechanisms, each in one place:
4
4
 
5
5
  1. Vocabulary (this module): `RolloutError` and the flat boundary types below. Each names the
6
6
  boundary a failure crossed — provider, harness, toolset, sandbox, task, or
7
- interception — so a recorded `trace.error.type` says where the rollout broke.
7
+ interception — so a recorded `trace.last_error.type` says where the rollout broke.
8
8
  2. Classification (`boundary`): the one helper that runs a framework→code boundary and attributes
9
9
  any escaping error to that boundary's type. Extension code (task hooks, harness subclasses)
10
10
  raises plain Python errors — it never constructs a `vf` error type; `boundary` classifies them.
@@ -104,10 +104,9 @@ class GEPAAdapter:
104
104
  "completion": trace.last_reply,
105
105
  "reward": trace.reward,
106
106
  }
107
- if trace.agent_name:
108
- record["agent"] = trace.agent_name
107
+ record["agent"] = trace.agent.name
109
108
  if trace.has_error:
110
- record["error"] = str(trace.error)
109
+ record["error"] = str(trace.last_error)
111
110
  if trace.stop_condition:
112
111
  record["stop_condition"] = trace.stop_condition
113
112
  for column in self.reflection_columns:
verifiers/v1/legacy.py CHANGED
@@ -24,6 +24,7 @@ from pydantic import ValidationError
24
24
 
25
25
  from verifiers.v1 import graph
26
26
  from verifiers.v1.clients.config import ClientConfig, TrainClientConfig
27
+ from verifiers.v1.configs.agent import AgentConfig
27
28
  from verifiers.v1.episode import Episode
28
29
  from verifiers.v1.serve.server import EnvServer
29
30
  from verifiers.v1.serve.types import (
@@ -34,6 +35,7 @@ from verifiers.v1.serve.types import (
34
35
  )
35
36
  from verifiers.v1.task import WireTaskData
36
37
  from verifiers.v1.trace import (
38
+ AgentInfo,
37
39
  Error,
38
40
  GenerationSpan,
39
41
  ModelCall,
@@ -227,7 +229,6 @@ def _timing(raw: Any) -> Timing:
227
229
  _V0_TO_V1_TRUNCATION_STOP = {
228
230
  "max_turns_reached": "max_turns",
229
231
  "prompt_too_long": "context_length",
230
- "timeout_reached": "harness_timeout",
231
232
  "max_total_completion_tokens_reached": "max_output_tokens",
232
233
  }
233
234
 
@@ -270,7 +271,10 @@ def rollout_output_to_trace(out: dict, task_idx: int) -> Trace:
270
271
  type="Task",
271
272
  data=_to_wire_task(task_idx, out.get("prompt"), out.get("answer")),
272
273
  ),
273
- tools=_to_v1_tools(out.get("tool_defs")),
274
+ # v0 rollouts carry no agent config — record the default, like the
275
+ # base task type above.
276
+ agent=AgentInfo(config=AgentConfig()),
277
+ tools=_to_v1_tools(out.get("tool_defs")) or [],
274
278
  rewards={"reward": Reward(score=float(out.get("reward") or 0.0))},
275
279
  metrics={k: float(v) for k, v in (out.get("metrics") or {}).items()},
276
280
  info=dict(out.get("info") or {}),
verifiers/v1/push.py CHANGED
@@ -58,8 +58,8 @@ def trace_to_sample(
58
58
  "example_id": trace.task.data.idx,
59
59
  "rollout_number": rollout_number,
60
60
  "episode_id": episode_id,
61
- "agent": trace.agent_name,
62
- "trainable": trace.trainable,
61
+ "agent": trace.agent.name,
62
+ "trainable": trace.agent.trainable,
63
63
  "task": task,
64
64
  "prompt": [],
65
65
  "completion": dump(branches[-1].messages) if branches else [],
@@ -73,8 +73,8 @@ def trace_to_sample(
73
73
  "is_completed": trace.is_completed,
74
74
  "is_truncated": trace.is_truncated,
75
75
  "metrics": trace.metrics,
76
- "error": trace.error.model_dump(mode="json", exclude_none=True)
77
- if trace.error
76
+ "error": trace.last_error.model_dump(mode="json", exclude_none=True)
77
+ if trace.last_error
78
78
  else None,
79
79
  "stop_condition": trace.stop_condition,
80
80
  "trajectory": [
@@ -124,7 +124,7 @@ def _run_metrics(episodes: list[Episode], traces: list[Trace]) -> dict[str, Any]
124
124
  back to all traces when none are trainable (same rule as the dashboard).
125
125
  `avg_error` is the share of EPISODES that aren't ok: a hook failure counts
126
126
  even when its traces are clean or it left none."""
127
- scored = [t for t in traces if t.trainable] or traces
127
+ scored = [t for t in traces if t.agent.trainable] or traces
128
128
  sums: dict[str, float] = {}
129
129
  counts: dict[str, int] = {}
130
130
  for trace in scored:
verifiers/v1/retries.py CHANGED
@@ -117,7 +117,9 @@ async def run_episode_with_retry(
117
117
  final = await run()
118
118
  if attempt == retry.max_retries or not episode_should_retry(final, retry):
119
119
  break
120
- cause = final.error or next((t.error for t in final.traces if t.error), None)
120
+ cause = final.error or next(
121
+ (t.last_error for t in final.traces if t.last_error), None
122
+ )
121
123
  history.extend(final.errors)
122
124
  for trace in final.traces:
123
125
  history.extend(trace.errors)
verifiers/v1/rollout.py CHANGED
@@ -24,7 +24,6 @@ import time
24
24
  from collections.abc import AsyncIterator, Callable
25
25
  from contextlib import AsyncExitStack, asynccontextmanager
26
26
 
27
- from verifiers import __version__
28
27
  from verifiers.v1.clients import ModelContext
29
28
  from verifiers.v1.configs.agent import AgentConfig
30
29
  from verifiers.v1.decorators import discover_decorated, invoke
@@ -52,9 +51,8 @@ from verifiers.v1.runtimes import (
52
51
  from verifiers.v1.session import RolloutLimits, RolloutSession
53
52
  from verifiers.v1.state import state_cls
54
53
  from verifiers.v1.task import Task, TaskData
55
- from verifiers.v1.trace import AgentInfo, Trace, TraceTask, VersionInfo
54
+ from verifiers.v1.trace import AgentInfo, Trace, TraceTask
56
55
  from verifiers.v1.types import Messages
57
- from verifiers.v1.utils.version import verifiers_commit
58
56
 
59
57
  logger = logging.getLogger(__name__)
60
58
 
@@ -118,7 +116,7 @@ class RolloutRun:
118
116
  wire_data: TaskData | None = None,
119
117
  has_user: bool = False,
120
118
  setup_timeout: float | None = None,
121
- harness_timeout: float | None = None,
119
+ agent_timeout: float | None = None,
122
120
  finalize_timeout: float | None = None,
123
121
  scoring_timeout: float | None = None,
124
122
  limits: RolloutLimits | None = None,
@@ -133,7 +131,8 @@ class RolloutRun:
133
131
  self.runtime_config = runtime_config
134
132
  self._has_user = has_user
135
133
  self._setup_timeout = setup_timeout
136
- self._harness_time_remaining = harness_timeout
134
+ self._agent_timeout = agent_timeout
135
+ self._agent_time_remaining = agent_timeout
137
136
  self._finalize_timeout = finalize_timeout
138
137
  self._scoring_timeout = scoring_timeout
139
138
  self._shared_tools = shared_tools or {}
@@ -146,7 +145,6 @@ class RolloutRun:
146
145
  data=task.data if wire_data is None else wire_data,
147
146
  ),
148
147
  state=state_cls(type(task))(),
149
- verifiers=VersionInfo(version=__version__, commit=verifiers_commit()),
150
148
  # The seat's resolved config, role overrides included — the agent
151
149
  # this trace can be reproduced with.
152
150
  agent=AgentInfo(config=agent_config),
@@ -166,7 +164,7 @@ class RolloutRun:
166
164
  self.deadline_at: float | None = None
167
165
  """The active harness segment's absolute deadline (event-loop clock), or
168
166
  None between segments / when unbounded. An interaction spends one cumulative
169
- `harness_timeout` budget only while its own segments run, so time awaiting
167
+ `agent_timeout` budget only while its own segments run, so time awaiting
170
168
  the caller (including another interleaved agent) cannot starve it."""
171
169
 
172
170
  @property
@@ -201,7 +199,7 @@ class RolloutRun:
201
199
  logger.exception("unexpected error in rollout %s", self.trace.id)
202
200
  self._failed = True
203
201
  self._failure = error
204
- self.trace.capture_error(error)
202
+ self.trace.record_error(error)
205
203
 
206
204
  async def open(self) -> bool:
207
205
  """Boot the rollout's world up to the point where segments can run: start
@@ -308,7 +306,8 @@ class RolloutRun:
308
306
  for an exchange the user opens, this is also the first segment, on an
309
307
  empty conversation); without, it launches on the task's own prompt.
310
308
  Returns whether the exchange can continue — a refused turn (limit, @stop),
311
- a timeout, a failure, or a segment that made no progress all end it."""
309
+ a failure (an expired agent timeout included), or a segment that made no
310
+ progress all end it."""
312
311
  if not self._opened or self._closed or not self.ok:
313
312
  return False
314
313
  trace = self.trace
@@ -317,11 +316,10 @@ class RolloutRun:
317
316
  segment_start = loop.time()
318
317
  self.deadline_at = (
319
318
  None
320
- if self._harness_time_remaining is None
321
- else segment_start + max(0.0, self._harness_time_remaining)
319
+ if self._agent_time_remaining is None
320
+ else segment_start + max(0.0, self._agent_time_remaining)
322
321
  )
323
322
  # Prefer an intercepted model/tool error to the harness exit it caused.
324
- # A timeout still scores the partial trajectory.
325
323
  try:
326
324
  async with asyncio.timeout_at(self.deadline_at):
327
325
  await self.harness.run(
@@ -335,11 +333,16 @@ class RolloutRun:
335
333
  messages,
336
334
  )
337
335
  except TimeoutError as e:
338
- # Only the rollout deadline reads as a clean truncation; a TimeoutError
339
- # from the harness's own I/O with no expired deadline is a failure —
340
- # recording it as a stop would score a broken run as a partial success.
336
+ # An expired rollout deadline is the agent breaking its time budget —
337
+ # an agent failure, never a clean stop. A TimeoutError from the
338
+ # harness's own I/O with no expired deadline stays the raw failure.
341
339
  if self.deadline_at is not None and (loop.time() >= self.deadline_at):
342
- trace.stop("harness_timeout")
340
+ self.fail(
341
+ HarnessError(
342
+ f"agent timeout: rollout exceeded its "
343
+ f"{self._agent_timeout:g}s budget"
344
+ )
345
+ )
343
346
  else:
344
347
  self.fail(e)
345
348
  return False
@@ -352,9 +355,9 @@ class RolloutRun:
352
355
  self.fail(e)
353
356
  return False
354
357
  finally:
355
- if self._harness_time_remaining is not None:
356
- self._harness_time_remaining = max(
357
- 0.0, self._harness_time_remaining - (loop.time() - segment_start)
358
+ if self._agent_time_remaining is not None:
359
+ self._agent_time_remaining = max(
360
+ 0.0, self._agent_time_remaining - (loop.time() - segment_start)
358
361
  )
359
362
  self.deadline_at = None
360
363
  if self._session.error is not None:
@@ -457,6 +460,6 @@ class RolloutRun:
457
460
  self.task.data.idx,
458
461
  trace.reward,
459
462
  trace.num_turns,
460
- trace.error.type if trace.error else trace.stop_condition,
463
+ trace.last_error.type if trace.last_error else trace.stop_condition,
461
464
  )
462
465
  return trace
@@ -62,7 +62,7 @@ class SubprocessRuntime(Runtime):
62
62
  stdout, stderr = await proc.communicate()
63
63
  finally:
64
64
  # If the await didn't finish, the caller cancelled it (e.g. the rollout's
65
- # scoring_timeout / harness_timeout fired): communicate() leaves the process
65
+ # scoring_timeout / agent_timeout fired): communicate() leaves the process
66
66
  # running, so SIGKILL its whole group (start_new_session => pgid == pid) — otherwise
67
67
  # a hung child (a wedged uv/sympy verify) outlives the rollout and leaks CPU. A
68
68
  # no-op once it has exited on its own.
verifiers/v1/task.py CHANGED
@@ -89,7 +89,7 @@ class TaskTimeout(StrictBaseModel):
89
89
  model_config = ConfigDict(frozen=True)
90
90
 
91
91
  setup: float | None = None
92
- harness: float | None = None
92
+ agent: float | None = None
93
93
  finalize: float | None = None
94
94
  scoring: float | None = None
95
95
 
@@ -260,9 +260,9 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
260
260
  if not authors and meta.get("author_name"):
261
261
  authors = [Author(name=meta["author_name"], email=meta.get("author_email"))]
262
262
  if harbor_config.ignore_timeouts:
263
- harness_timeout = scoring_timeout = None
263
+ agent_timeout = scoring_timeout = None
264
264
  else:
265
- harness_timeout = (
265
+ agent_timeout = (
266
266
  parsed.agent.timeout_sec
267
267
  if "timeout_sec" in parsed.agent.model_fields_set
268
268
  else None
@@ -290,8 +290,8 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
290
290
  else list(network.allowed_hosts)
291
291
  ),
292
292
  timeout=TaskTimeout(
293
- harness=harness_timeout * harbor_config.timeout_multiplier
294
- if harness_timeout is not None
293
+ agent=agent_timeout * harbor_config.timeout_multiplier
294
+ if agent_timeout is not None
295
295
  else None,
296
296
  scoring=scoring_timeout * harbor_config.timeout_multiplier
297
297
  if scoring_timeout is not None