verifiers 0.2.2.dev53__py3-none-any.whl → 0.2.2.dev55__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
verifiers/v1/__init__.py CHANGED
@@ -115,10 +115,10 @@ from verifiers.v1.taskset import Taskset
115
115
  from verifiers.v1.trace import (
116
116
  TRACE_VERSION,
117
117
  AgentInfo,
118
+ AgentSpan,
118
119
  Branch,
119
120
  Error,
120
121
  EvalRunInfo,
121
- GenerationSpan,
122
122
  ModelCall,
123
123
  Reward,
124
124
  RunInfo,
@@ -144,7 +144,6 @@ from verifiers.v1.types import (
144
144
  Response,
145
145
  Sampling,
146
146
  SamplingConfig,
147
- StrictBaseModel,
148
147
  SystemMessage,
149
148
  TextContentPart,
150
149
  Tool,
@@ -177,7 +176,6 @@ __all__ = [ # noqa: RUF022 - grouped by public API area
177
176
  "Response",
178
177
  "Sampling",
179
178
  "SamplingConfig",
180
- "StrictBaseModel",
181
179
  "SystemMessage",
182
180
  "TextContentPart",
183
181
  "Tool",
@@ -213,7 +211,7 @@ __all__ = [ # noqa: RUF022 - grouped by public API area
213
211
  "Timing",
214
212
  "TimeSpan",
215
213
  "TimeSplit",
216
- "GenerationSpan",
214
+ "AgentSpan",
217
215
  "Error",
218
216
  # decorators
219
217
  "stop",
verifiers/v1/artifacts.py CHANGED
@@ -8,9 +8,7 @@ import uuid
8
8
  from pathlib import PurePosixPath
9
9
  from typing import TYPE_CHECKING
10
10
 
11
- from pydantic import Field
12
-
13
- from verifiers.v1.types import StrictBaseModel
11
+ from pydantic import BaseModel, Field
14
12
 
15
13
  if TYPE_CHECKING:
16
14
  from verifiers.v1.runtimes import Runtime
@@ -25,7 +23,7 @@ MAX_ARTIFACT_BYTES = 32 * 1024 * 1024
25
23
  agent's image, so the repo is already there and only its output has to travel."""
26
24
 
27
25
 
28
- class Artifact(StrictBaseModel):
26
+ class Artifact(BaseModel):
29
27
  """One path to restore at the same location in another runtime."""
30
28
 
31
29
  source: str
@@ -393,18 +393,19 @@ def _breakdown(scored: list[Trace], done: list[Trace]) -> Table | None:
393
393
  phase_count: dict[str, int] = {}
394
394
  model_secs = harness_secs = 0.0
395
395
  for trace in done:
396
- prompt, completion, cached, reasoning, _ = _tokens(trace)
397
- total_in += prompt
398
- total_out += completion
399
- if cached is not None:
400
- total_cached += cached
401
- have_cached = True
402
- if reasoning is not None:
403
- total_reasoning += reasoning
404
- have_reasoning = True
405
- if trace.usage is not None and trace.usage.cost is not None:
406
- total_cost += trace.usage.cost
407
- have_cost = True
396
+ total_in += trace.num_input_tokens
397
+ total_out += trace.num_output_tokens
398
+ usage = trace.usage
399
+ if usage is not None:
400
+ if usage.cached_input_tokens is not None:
401
+ total_cached += usage.cached_input_tokens
402
+ have_cached = True
403
+ if usage.reasoning_tokens is not None:
404
+ total_reasoning += usage.reasoning_tokens
405
+ have_reasoning = True
406
+ if usage.cost is not None:
407
+ total_cost += usage.cost
408
+ have_cost = True
408
409
  # Judge / auxiliary scoring calls (off the message graph) shown separately from the agent's.
409
410
  judge = Usage.aggregate(trace.extra_usage)
410
411
  if judge is not None:
@@ -413,13 +414,13 @@ def _breakdown(scored: list[Trace], done: list[Trace]) -> Table | None:
413
414
  if judge.cost is not None:
414
415
  total_judge_cost += judge.cost
415
416
  have_judge = True
416
- for phase in ("boot", "setup", "generation", "finalize", "scoring"):
417
+ for phase in ("boot", "setup", "agent", "finalize", "scoring"):
417
418
  span = getattr(trace.timing, phase)
418
419
  if span.end: # phase was timed for this rollout
419
420
  phase_secs[phase] = phase_secs.get(phase, 0.0) + span.duration
420
421
  phase_count[phase] = phase_count.get(phase, 0) + 1
421
- model_secs += trace.timing.generation.model.duration
422
- harness_secs += trace.timing.generation.harness.duration
422
+ model_secs += trace.timing.agent.model.duration
423
+ harness_secs += trace.timing.agent.harness.duration
423
424
  if (
424
425
  total_in
425
426
  or total_out
@@ -448,12 +449,12 @@ def _breakdown(scored: list[Trace], done: list[Trace]) -> Table | None:
448
449
  usage.append(cost)
449
450
  grid.add_row("usage", " · ".join(usage))
450
451
  time_segments = []
451
- for phase in ("boot", "setup", "generation", "finalize", "scoring"):
452
+ for phase in ("boot", "setup", "agent", "finalize", "scoring"):
452
453
  count = phase_count.get(phase)
453
454
  if not count:
454
455
  continue
455
456
  segment = f"{phase} {format_time(phase_secs[phase] / count)}"
456
- if phase == "generation":
457
+ if phase == "agent":
457
458
  segment += (
458
459
  f" (model {format_time(model_secs / count)}"
459
460
  f" + harness {format_time(harness_secs / count)})"
@@ -464,26 +465,6 @@ def _breakdown(scored: list[Trace], done: list[Trace]) -> Table | None:
464
465
  return grid if grid.row_count else None
465
466
 
466
467
 
467
- def _tokens(trace: Trace) -> tuple[int, int, int | None, int | None, int]:
468
- """Input/output tokens summed across all branches: per branch, output is every assistant
469
- (completion) token generated across its turns and input is the fed-in tokens counted once
470
- (system + user + tool) — the final sequence minus everything the model generated. A rollout
471
- yields one training sample per branch (a linear trace is a single branch; compaction and
472
- subagents add more), so the totals sum them — matching `Trace.num_input_tokens` /
473
- `Trace.num_output_tokens`, whose sum is `num_total_tokens`.
474
-
475
- Both counts come from provider-reported usage. Returns the branch count from the same derived
476
- view so each dashboard tick materializes it once."""
477
- usage = trace.usage
478
- cached = usage.cached_input_tokens if usage else None
479
- reasoning = usage.reasoning_tokens if usage else None
480
- branches = trace.branches
481
- nbranches = len(branches)
482
- prompt = sum(b.num_input_tokens for b in branches)
483
- completion = sum(b.num_output_tokens for b in branches)
484
- return prompt, completion, cached, reasoning, nbranches
485
-
486
-
487
468
  def _stage(trace: Trace) -> str:
488
469
  """The stage a live (not-yet-done) rollout is in, derived from its trace's timing
489
470
  spans — the engine opens and closes each span exactly at the stage transitions, so
@@ -495,7 +476,7 @@ def _stage(trace: Trace) -> str:
495
476
  for stage, span in (
496
477
  ("scoring", trace.timing.scoring),
497
478
  ("finalize", trace.timing.finalize),
498
- ("running", trace.timing.generation),
479
+ ("running", trace.timing.agent),
499
480
  ("setup", trace.timing.setup),
500
481
  ("boot", trace.timing.boot),
501
482
  ):
@@ -551,7 +532,9 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
551
532
  base = f"name={task.name[:32]}" if task.name else f"idx={task.idx}"
552
533
  if not slot.traces:
553
534
  if slot.done: # the env's rollout() itself failed before any trace
554
- error = slot.episode.error if slot.episode is not None else None
535
+ error = (
536
+ slot.episode.last_error if slot.episode is not None else None
537
+ )
555
538
  group_rows.append(
556
539
  (
557
540
  "error",
@@ -602,7 +585,7 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
602
585
  end = (
603
586
  t.timing.scoring.end
604
587
  or t.timing.finalize.end
605
- or t.timing.generation.end
588
+ or t.timing.agent.end
606
589
  # a rollout that errored in boot/setup has only that span's end — freeze there
607
590
  # once done, else (still running) the timer would grow off `now` forever
608
591
  or (
@@ -612,8 +595,12 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
612
595
  )
613
596
  or now
614
597
  )
615
- prompt, completion, cached, reasoning, nbranches = _tokens(t)
616
- cost = t.usage.cost if t.usage else None
598
+ prompt, completion = t.num_input_tokens, t.num_output_tokens
599
+ nbranches = t.num_branches
600
+ usage = t.usage
601
+ cached = usage.cached_input_tokens if usage else None
602
+ reasoning = usage.reasoning_tokens if usage else None
603
+ cost = usage.cost if usage else None
617
604
  tokens = ""
618
605
  if prompt or completion:
619
606
  tokens = f"{format_count(prompt)}/{format_count(completion)} tokens"
verifiers/v1/cli/debug.py CHANGED
@@ -123,16 +123,16 @@ def record_debug_error(
123
123
  action_timeout: float | None,
124
124
  ) -> None:
125
125
  now = time.time()
126
- for span in (trace.timing.boot, trace.timing.setup, trace.timing.generation):
126
+ for span in (trace.timing.boot, trace.timing.setup, trace.timing.agent):
127
127
  if span.start and not span.end:
128
128
  span.end = now
129
- in_action = bool(trace.timing.generation.start)
129
+ in_action = bool(trace.timing.agent.start)
130
130
  stage = (
131
131
  "debug action" if in_action else "setup" if trace.timing.setup.start else "boot"
132
132
  )
133
133
  timeout = action_timeout if in_action else setup_timeout
134
134
  error_start = (
135
- trace.timing.generation.start
135
+ trace.timing.agent.start
136
136
  if in_action
137
137
  else trace.timing.setup.start or trace.timing.boot.start
138
138
  )
@@ -234,9 +234,9 @@ async def debug_task(task: Task, config: DebugConfig) -> tuple[Trace, bool]:
234
234
  await runtime.prepare_execution([])
235
235
  trace.timing.setup.end = time.time()
236
236
 
237
- trace.timing.generation.start = time.time()
237
+ trace.timing.agent.start = time.time()
238
238
  debug.update(await run_action(runtime, config))
239
- trace.timing.generation.end = time.time()
239
+ trace.timing.agent.end = time.time()
240
240
  if not debug.get("ok"):
241
241
  record_action_failure(trace, debug)
242
242
  trace.stop(str(debug["reason"]))
@@ -246,7 +246,7 @@ async def debug_task(task: Task, config: DebugConfig) -> tuple[Trace, bool]:
246
246
  except Exception as e: # noqa: BLE001 - persist any framework failure on the trace
247
247
  record_debug_error(trace, debug, e, setup_timeout, config.timeout.total)
248
248
  finally:
249
- trace.split_generation()
249
+ trace.split_agent_time()
250
250
  trace.info["debug"] = debug
251
251
  try:
252
252
  await runtime.stop()
@@ -31,7 +31,7 @@ from verifiers.v1.cli.resolve import narrow_taskset_config
31
31
  from verifiers.v1.configs.agent import WireAgentConfig
32
32
  from verifiers.v1.configs.cli.replay import ReplayConfig
33
33
  from verifiers.v1.state import state_cls
34
- from verifiers.v1.task import Task, WireTaskData, task_data_cls
34
+ from verifiers.v1.task import Task, WireTaskData
35
35
  from verifiers.v1.trace import Trace
36
36
  from verifiers.v1.utils.interrupt import install_interrupt
37
37
  from verifiers.v1.utils.logging import setup_logging
@@ -76,7 +76,7 @@ async def run_replay(config: ReplayConfig, source: Path, out: Path) -> list[Trac
76
76
  "score() — multi-agent runs don't support replay"
77
77
  )
78
78
  task_cls = vf.task_type(config.taskset.id)
79
- data_cls = task_data_cls(task_cls)
79
+ data_cls = task_cls.data_type()
80
80
  # `WireTaskData` reads any taskset's saved task without importing its Task type.
81
81
  # An episode may hold no traces (its env hooks failed before any agent ran);
82
82
  # there's nothing to re-score, so it drops out in the flatten. Each kept trace
@@ -84,7 +84,7 @@ async def run_replay(config: ReplayConfig, source: Path, out: Path) -> list[Trac
84
84
  episodes = read_episodes(
85
85
  source, Trace[WireTaskData, state_cls(task_cls), WireAgentConfig]
86
86
  )
87
- sourced = [(trace, e.env) for e in episodes for trace in e.traces]
87
+ sourced = [(trace, e.env.id) for e in episodes for trace in e.traces]
88
88
  if config.num_traces is not None:
89
89
  sourced = sourced[: config.num_traces]
90
90
  traces = [trace for trace, _ in sourced]
verifiers/v1/env.py CHANGED
@@ -20,7 +20,7 @@ from verifiers.v1.configs.env import (
20
20
  _declared_agent_configs,
21
21
  default_agent_harness,
22
22
  )
23
- from verifiers.v1.episode import Episode
23
+ from verifiers.v1.episode import EnvInfo, Episode
24
24
  from verifiers.v1.errors import EnvError, boundary
25
25
  from verifiers.v1.harness import Harness, HarnessConfig
26
26
  from verifiers.v1.interception import (
@@ -257,7 +257,7 @@ class Env(ABC, Generic[ConfigT]):
257
257
  completed subset, its exception on the episode's `errors`. `on_trace` observes
258
258
  each agent-run's trace at mint; `on_discard` its abandonment (a per-agent
259
259
  retry mints a replacement)."""
260
- episode = Episode(env=self.config.env_id)
260
+ episode = Episode(env=EnvInfo(id=self.config.env_id))
261
261
  agents = self._episode_agents(ctx, episode.traces, on_trace, on_discard)
262
262
  try:
263
263
  async with asyncio.timeout(self.config.timeout.episode):
@@ -19,10 +19,9 @@ import re
19
19
  import tomllib
20
20
  from pathlib import Path
21
21
 
22
- from pydantic import Field, field_validator
22
+ from pydantic import BaseModel, Field, field_validator
23
23
 
24
24
  import verifiers.v1 as vf
25
- from verifiers.v1.types import StrictBaseModel
26
25
  from verifiers.v1.utils.compile import validate_pairing
27
26
 
28
27
  VERDICT_FILE = "/tmp/verdict.json"
@@ -39,7 +38,7 @@ TASK_SECTION = """\
39
38
  {prompt}"""
40
39
 
41
40
 
42
- class Criterion(StrictBaseModel):
41
+ class Criterion(BaseModel):
43
42
  """One rubric criterion — the plugged rubric judge's format, mirrored so the
44
43
  same `criteria` files grade both judges."""
45
44
 
verifiers/v1/episode.py CHANGED
@@ -3,51 +3,79 @@
3
3
  import uuid
4
4
  from typing import Generic
5
5
 
6
- from pydantic import Field
6
+ from pydantic import BaseModel, Field
7
7
 
8
8
  from verifiers.v1.configs.agent import WireAgentConfig
9
9
  from verifiers.v1.state import State, StateT
10
10
  from verifiers.v1.task import DataT, WireTaskData
11
11
  from verifiers.v1.trace import AgentConfigT, Error, Trace
12
- from verifiers.v1.types import StrictBaseModel
12
+ from verifiers.v1.types import Usage
13
13
 
14
14
 
15
- class Episode(StrictBaseModel, Generic[DataT, StateT, AgentConfigT]):
16
- """One run of a task, whole: its identity and standing (`id`, `env`, `errors`)
17
- next to its flat `traces` — the object `finalize()` receives, the engine
18
- returns, and the durability envelope: one episode is one `traces.jsonl` line
19
- and one serve reply, so it persists and arrives whole or not at all — a torn
20
- line is the whole episode owed again, and a failure before any trace minted
21
- still leaves its errors here. Episode standing lives ONLY here (zero
22
- redundancy on the traces); per-trace facts (`agent`, per-trace errors) stay
23
- on the traces, which remain the atomic unit.
15
+ class EnvInfo(BaseModel):
16
+ """The env that ran the episode, self-describing without the run's config."""
24
17
 
25
- `errors` are failures not attributable to any one trace (the env's
26
- `run`/`finalize` hooks, plus prior attempts' when retried).
18
+ id: str = ""
19
+ """`EnvConfig.env_id`, e.g. `agentic-judge+gsm8k-v1`."""
27
20
 
28
- The type parameters serve the wire loaders: `WireEpisode` reads any taskset's
29
- episodes without importing the taskset."""
21
+
22
+ class Episode(BaseModel, Generic[DataT, StateT, AgentConfigT]):
23
+ """The artifact Env.run produces. Contains multiple agents' traces."""
30
24
 
31
25
  id: str = Field(default_factory=lambda: uuid.uuid4().hex)
32
- env: str = ""
33
- """The env that ran the episode (`EnvConfig.env_id`, e.g.
34
- `agentic-judge+gsm8k-v1`)."""
26
+
27
+ env: EnvInfo = Field(default_factory=EnvInfo)
28
+ """The env that produced this episode."""
35
29
  ok: bool = False
36
- """THE success sentinel — the resume unit's keep-verdict, stamped by the
37
- engine when the final attempt's hooks and every trace concluded clean.
38
- Distinct from `errors` emptiness: a retried-and-recovered episode is `ok`
39
- and still keeps its earlier attempts' errors."""
30
+ """Whether the episode completed successfully."""
40
31
  errors: list[Error] = Field(default_factory=list)
32
+ """Every error captured across attempts, oldest to newest."""
41
33
  traces: list[Trace[DataT, StateT, AgentConfigT]] = Field(default_factory=list)
34
+ """Every agent's trace, in completion order."""
42
35
 
43
36
  @property
44
- def error(self) -> Error | None:
37
+ def last_error(self) -> Error | None:
38
+ """The last episode-level error captured across attempts."""
45
39
  return self.errors[-1] if self.errors else None
46
40
 
41
+ @property
42
+ def usage(self) -> Usage | None:
43
+ """Provider-reported usage summed across every trace's model calls;
44
+ judge/off-graph usage stays on the traces (`Trace.extra_usage`)."""
45
+ return Usage.aggregate(u for t in self.traces if (u := t.usage) is not None)
46
+
47
+ @property
48
+ def num_input_tokens(self) -> int:
49
+ """Fed-in tokens (system + user + tool), summed across traces."""
50
+ return sum(t.num_input_tokens for t in self.traces)
51
+
52
+ @property
53
+ def num_output_tokens(self) -> int:
54
+ """Model-generated tokens across all turns, summed across traces."""
55
+ return sum(t.num_output_tokens for t in self.traces)
56
+
57
+ @property
58
+ def num_total_tokens(self) -> int:
59
+ """Final sequence lengths per branch, summed across traces."""
60
+ return sum(t.num_total_tokens for t in self.traces)
61
+
62
+ @property
63
+ def num_turns(self) -> int:
64
+ """Sampled turns, summed across traces."""
65
+ return sum(t.num_turns for t in self.traces)
66
+
67
+ @property
68
+ def by_agent(self) -> dict[str, list[Trace[DataT, StateT, AgentConfigT]]]:
69
+ """Traces grouped by agent name (e.g. n solvers), in completion order."""
70
+ grouped: dict[str, list[Trace[DataT, StateT, AgentConfigT]]] = {}
71
+ for trace in self.traces:
72
+ grouped.setdefault(trace.agent.name, []).append(trace)
73
+ return grouped
74
+
47
75
  @classmethod
48
76
  def of(cls, trace: Trace, env: str = "") -> "Episode":
49
77
  """The single-agent record: one trace as its own episode."""
50
- return cls(env=env, traces=[trace], ok=trace.ok)
78
+ return cls(env=EnvInfo(id=env), traces=[trace], ok=trace.ok)
51
79
 
52
80
 
53
81
  WireEpisode = Episode[WireTaskData, State, WireAgentConfig]
verifiers/v1/graph.py CHANGED
@@ -25,7 +25,7 @@ from dataclasses import dataclass
25
25
  from typing import TYPE_CHECKING, Any
26
26
 
27
27
  import numpy as np
28
- from pydantic import ConfigDict, Field, field_serializer, field_validator
28
+ from pydantic import BaseModel, ConfigDict, Field, field_serializer, field_validator
29
29
  from pydantic.json_schema import SkipJsonSchema
30
30
  from renderers.base import MultiModalData, PlaceholderRange, RenderedTokens
31
31
 
@@ -34,7 +34,6 @@ from verifiers.v1.types import (
34
34
  KeptTokens,
35
35
  Message,
36
36
  Response,
37
- StrictBaseModel,
38
37
  TextContentPart,
39
38
  Tool,
40
39
  ToolMessage,
@@ -62,7 +61,7 @@ def _decode_ndarray(d: dict) -> np.ndarray:
62
61
  return np.frombuffer(d["data"], dtype=np.dtype(d["dtype"])).reshape(d["shape"])
63
62
 
64
63
 
65
- class MessageNode(StrictBaseModel):
64
+ class MessageNode(BaseModel):
66
65
  """One message in the graph: a message plus the tokens it adds to the cumulative
67
66
  sequence. Concatenating a root→leaf path's nodes reconstructs that branch's full token
68
67
  sequence; the mask/logprobs make it a training sample."""
@@ -121,7 +120,7 @@ class MessageNode(StrictBaseModel):
121
120
  sampling-replay training. Rides the wire as raw-bytes `__nd__` dicts; kept off disk
122
121
  by the dump-site `exclude` in prime-rl."""
123
122
 
124
- model_config = ConfigDict(extra="forbid", arbitrary_types_allowed=True)
123
+ model_config = ConfigDict(arbitrary_types_allowed=True)
125
124
 
126
125
  @field_serializer("multi_modal_data")
127
126
  def serialize_multi_modal_data(self, mmd: MultiModalData | None) -> dict | None:
verifiers/v1/judge.py CHANGED
@@ -62,7 +62,7 @@ from verifiers.v1.configs.judge import (
62
62
  )
63
63
  from verifiers.v1.dialects.chat import message_to_wire
64
64
  from verifiers.v1.scoring import parse_judge_choice
65
- from verifiers.v1.types import Messages, StrictBaseModel, Usage
65
+ from verifiers.v1.types import Messages, Usage
66
66
  from verifiers.v1.utils.generic import concrete_type
67
67
 
68
68
  if TYPE_CHECKING:
@@ -72,7 +72,7 @@ if TYPE_CHECKING:
72
72
  ParsedT = TypeVar("ParsedT")
73
73
 
74
74
 
75
- class JudgeResponse(StrictBaseModel, Generic[ParsedT]):
75
+ class JudgeResponse(BaseModel, Generic[ParsedT]):
76
76
  text: str
77
77
  parsed: ParsedT | None = None
78
78
  usage: Usage | None = None
@@ -197,8 +197,7 @@ class Judge(Generic[ParsedT, ConfigT]):
197
197
  )
198
198
  if response.parsed is None:
199
199
  raise RuntimeError(
200
- f"judge returned no parseable structured output "
201
- f"(finish_reason={choice.finish_reason})"
200
+ f"judge returned no parseable structured output (finish_reason={choice.finish_reason})"
202
201
  )
203
202
  else:
204
203
  completion = await client.chat.completions.create(**kwargs)
@@ -9,13 +9,13 @@ from functools import cached_property
9
9
  from pathlib import Path
10
10
  from typing import cast
11
11
 
12
- from pydantic import Field, field_validator
12
+ from pydantic import BaseModel, Field, field_validator
13
13
 
14
14
  from verifiers.v1.configs.judge import JudgeConfig
15
15
  from verifiers.v1.judge import Judge, JudgeView, judge_question, judge_response
16
16
  from verifiers.v1.task import TaskData
17
17
  from verifiers.v1.trace import Trace
18
- from verifiers.v1.types import ID, StrictBaseModel
18
+ from verifiers.v1.types import ID
19
19
 
20
20
  RUBRIC_PROMPT = (Path(__file__).resolve().parent / "rubric.txt").read_text(
21
21
  encoding="utf-8"
@@ -56,7 +56,7 @@ def first_verdicts_object(text: str) -> dict | None:
56
56
  return None
57
57
 
58
58
 
59
- class Criterion(StrictBaseModel):
59
+ class Criterion(BaseModel):
60
60
  name: str
61
61
  """Key for the criterion's metric (`<judge name>/<name>`) and its `weights` override."""
62
62
  text: str
@@ -109,13 +109,13 @@ class RubricJudgeConfig(JudgeConfig):
109
109
  handle either. Transient HTTP failures are already retried by the OpenAI client."""
110
110
 
111
111
 
112
- class CriterionVerdict(StrictBaseModel):
112
+ class CriterionVerdict(BaseModel):
113
113
  name: str
114
114
  reason: str
115
115
  verdict: str
116
116
 
117
117
 
118
- class RubricVerdicts(StrictBaseModel):
118
+ class RubricVerdicts(BaseModel):
119
119
  verdicts: list[CriterionVerdict]
120
120
 
121
121
 
verifiers/v1/legacy.py CHANGED
@@ -36,8 +36,8 @@ from verifiers.v1.serve.types import (
36
36
  from verifiers.v1.task import WireTaskData
37
37
  from verifiers.v1.trace import (
38
38
  AgentInfo,
39
+ AgentSpan,
39
40
  Error,
40
- GenerationSpan,
41
41
  ModelCall,
42
42
  Reward,
43
43
  TimeSpan,
@@ -199,7 +199,8 @@ def _to_v1_tokens(raw: Any) -> TurnTokens | None:
199
199
  def _timing(raw: Any) -> Timing:
200
200
  """Map the v0 timing record's generation/scoring durations onto a v1 ``Timing``
201
201
  (we only have durations, so each span is encoded as start=0, end=duration).
202
- v0's per-turn ``model``/``env`` span collections carry the generation split."""
202
+ v0's ``generation`` duration becomes the agent span; its per-turn ``model``/``env``
203
+ span collections carry the model/harness split."""
203
204
 
204
205
  def _dur(node: Any) -> float:
205
206
  if isinstance(node, dict):
@@ -212,7 +213,7 @@ def _timing(raw: Any) -> Timing:
212
213
 
213
214
  raw = raw or {}
214
215
  return Timing(
215
- generation=GenerationSpan(
216
+ agent=AgentSpan(
216
217
  start=0.0,
217
218
  end=_dur(raw.get("generation")),
218
219
  model=TimeSplit(duration=_dur(raw.get("model"))),
verifiers/v1/rollout.py CHANGED
@@ -297,7 +297,7 @@ class RolloutRun:
297
297
  raise
298
298
  now = time.time()
299
299
  self.trace.timing.setup.end = now
300
- self.trace.timing.generation.start = now
300
+ self.trace.timing.agent.start = now
301
301
  return True
302
302
 
303
303
  async def step(self, messages: Messages | None = None) -> bool:
@@ -397,8 +397,8 @@ class RolloutRun:
397
397
  try:
398
398
  await self._stack.aclose()
399
399
  finally:
400
- if trace.timing.generation.start and not trace.timing.generation.end:
401
- trace.timing.generation.end = time.time()
400
+ if trace.timing.agent.start and not trace.timing.agent.end:
401
+ trace.timing.agent.end = time.time()
402
402
  if not self._failed and self._opened:
403
403
  trace.timing.finalize.start = time.time()
404
404
  async with boundary(TaskError, "task finalize"):
@@ -430,13 +430,13 @@ class RolloutRun:
430
430
  for span in (
431
431
  trace.timing.boot,
432
432
  trace.timing.setup,
433
- trace.timing.generation,
433
+ trace.timing.agent,
434
434
  trace.timing.finalize,
435
435
  trace.timing.scoring,
436
436
  ):
437
437
  if span.start and not span.end:
438
438
  span.end = now
439
- trace.split_generation()
439
+ trace.split_agent_time()
440
440
  if runtime is not None:
441
441
  try:
442
442
  await self.harness.cleanup(trace, runtime)
@@ -22,7 +22,7 @@ from verifiers.v1.serve.types import (
22
22
  RunRequest,
23
23
  RunResponse,
24
24
  )
25
- from verifiers.v1.task import Task, task_data_cls
25
+ from verifiers.v1.task import Task
26
26
  from verifiers.v1.types import SamplingConfig
27
27
 
28
28
  logger = logging.getLogger(__name__)
@@ -39,7 +39,7 @@ class EnvServer:
39
39
  self.taskset_id = config.taskset.id
40
40
  self.env = load_environment(config)
41
41
  self.task_cls = type(self.env.taskset).task_type()
42
- self.data_cls = task_data_cls(self.task_cls)
42
+ self.data_cls = self.task_cls.data_type()
43
43
  # A dispatched task is its client-side model_dump(): a field excluded from
44
44
  # serialization would vanish on the wire and rebuild silently defaulted, so
45
45
  # refuse to serve such a taskset.
verifiers/v1/state.py CHANGED
@@ -4,14 +4,13 @@ Tool servers synchronize it through the interception state channel. It is exclud
4
4
  from serialized traces.
5
5
  """
6
6
 
7
- from pydantic import ConfigDict, Field
7
+ from pydantic import BaseModel, ConfigDict, Field
8
8
  from typing_extensions import TypeVar
9
9
 
10
- from verifiers.v1.types import StrictBaseModel
11
10
  from verifiers.v1.utils.generic import concrete_type
12
11
 
13
12
 
14
- class State(StrictBaseModel):
13
+ class State(BaseModel):
15
14
  model_config = ConfigDict(ser_json_inf_nan="constants")
16
15
  artifacts: dict[str, bytes] = Field(default_factory=dict)
17
16
 
verifiers/v1/task.py CHANGED
@@ -1,25 +1,10 @@
1
- """Task data, configuration, behavior, and scoring.
2
-
3
- `TaskData` is the wire half: a frozen pydantic model carrying everything a rollout's
4
- row IS — the base fields plus your typed, task-specific fields. It rides on
5
- `trace.task.data`, is what `traces.jsonl` stores, and what tool servers receive
6
- over the `/task` channel. Subclass it per dataset.
7
-
8
- `Task` is the behavior half: runtime prep (`setup`/`finalize`), server declarations
9
- (`tools`), well-formedness (`validate`), and per-trace judgement
10
- (`@reward`/`@metric` methods plus the plugged judges from `config.judges`, run by
11
- `score`). Subclass per dataset and parameterize `Task[MyData, MyState, MyConfig]`
12
- (all three default); judgement that compares sibling traces lives on
13
- `Env.finalize` instead.
14
-
15
- A Task instance is shared across its rollouts (`-r n` runs hold the same instance),
16
- so hooks must not stash per-rollout state on `self` — that lives on the trace
17
- (`trace.state`).
18
-
19
- On the wire only the data travels (plus the producing class's name,
20
- `trace.task.type`): a saved row reads back as `WireTaskData` without importing the
21
- taskset; a re-scoring consumer (`replay`) rebuilds the declared `TaskData` type and
22
- wraps it in the declared `Task` — one task type per taskset.
1
+ """Task data + behavior.
2
+
3
+ `TaskData` is the wire half: a frozen pydantic model carrying the data which
4
+ initializes a task instance. Rides on `trace.task.data` in `traces.jsonl`.
5
+
6
+ `Task` is the behavior half: runtime prep (`setup`/`finalize`), tool declarations
7
+ (`tools`), and scoring (`@reward`/`@metric`) methods.
23
8
  """
24
9
 
25
10
  from __future__ import annotations
@@ -30,7 +15,7 @@ import logging
30
15
  from collections.abc import Mapping
31
16
  from typing import TYPE_CHECKING, ClassVar, Generic, Self
32
17
 
33
- from pydantic import ConfigDict, Field
18
+ from pydantic import BaseModel, ConfigDict, Field
34
19
  from pydantic_config import BaseConfig
35
20
  from typing_extensions import TypeVar
36
21
 
@@ -39,7 +24,7 @@ from verifiers.v1.configs.task import TaskConfig
39
24
  from verifiers.v1.decorators import discover_decorated, invoke_all
40
25
  from verifiers.v1.errors import TaskError, boundary
41
26
  from verifiers.v1.state import StateT
42
- from verifiers.v1.types import Messages, StrictBaseModel, content_text
27
+ from verifiers.v1.types import Messages, content_text
43
28
  from verifiers.v1.utils.generic import concrete_type
44
29
 
45
30
  if TYPE_CHECKING:
@@ -51,7 +36,9 @@ if TYPE_CHECKING:
51
36
  logger = logging.getLogger(__name__)
52
37
 
53
38
 
54
- class TaskResources(StrictBaseModel):
39
+ class TaskResources(BaseModel):
40
+ """Optional resource limits for the task."""
41
+
55
42
  model_config = ConfigDict(frozen=True)
56
43
 
57
44
  cpu: float | None = None
@@ -64,37 +51,41 @@ class TaskResources(StrictBaseModel):
64
51
  """Disk in GB (enforced by prime; advisory on docker/modal)."""
65
52
 
66
53
 
67
- class TaskTimeout(StrictBaseModel):
54
+ class TaskTimeout(BaseModel):
68
55
  """Optional per-task timeout overrides, in seconds."""
69
56
 
70
57
  model_config = ConfigDict(frozen=True)
71
58
 
72
59
  setup: float | None = None
60
+ """Timeout (in seconds) for the task's setup hook."""
73
61
  agent: float | None = None
62
+ """Timeout (in seconds) for the agent's solve attempt."""
74
63
  finalize: float | None = None
64
+ """Timeout (in seconds) for the task's finalize hook."""
75
65
  scoring: float | None = None
66
+ """Timeout (in seconds) for the task's scoring."""
76
67
 
77
68
 
78
- class TaskData(StrictBaseModel):
79
- """The task's wire half: one row's pure data, a frozen pydantic model. Subclass
80
- per dataset to add typed task-specific fields next to the base fields; behavior
81
- lives on `Task`, which wraps this (`self.data`)."""
82
-
69
+ class TaskData(BaseModel):
83
70
  model_config = ConfigDict(frozen=True)
84
71
 
85
72
  idx: int | None = None
86
- """Dataset row index — set by tasksets (selection, resume, and grouping key
87
- on the eval path); `None` for ad-hoc tasks minted in scripts."""
73
+ """Taskset-autogenerated index."""
88
74
  name: str | None = None
75
+ """Optional human-readable task name."""
89
76
  description: str | None = None
77
+ """Optional human-readable task description."""
78
+
90
79
  prompt: str | Messages | None = None
91
- """Initial user prompt; `None` means the user opens the conversation — run the
92
- task through `agent.interaction()`, whose first `turn(message)` speaks first. (A
93
- default, not just optional: the wire drops `None`s — `traces.jsonl` rows for
94
- prompt-less tasks must read back.)"""
80
+ """Initial user prompt; unset if the user opens the conversation."""
95
81
  system_prompt: str | None = None
82
+ """Optional system prompt to prepend to the user prompt."""
83
+
96
84
  image: str | None = None
85
+ """Optional Docker image to use for the task. Only relevant for tasks that run in a container."""
97
86
  workdir: str | None = None
87
+ """Optional working directory to use for the task. Only relevant for tasks that run in a container."""
88
+
98
89
  network_allow: list[str] = Field(default_factory=lambda: ["*"])
99
90
  """Execution-time destinations requested by this task. `*` leaves the runtime
100
91
  allowlist unchanged; a concrete list replaces a wildcard or combines with existing
@@ -103,11 +94,13 @@ class TaskData(StrictBaseModel):
103
94
  """Execution-time destinations denied by this task and combined with runtime
104
95
  blocks. Non-empty concrete allowlists cannot be combined with blocklists. Docker
105
96
  framework routes take precedence; ordinary Prime deny rules pass through unchanged."""
97
+
106
98
  artifacts: list[Artifact] = Field(default_factory=list)
107
99
  """Paths collected from one runtime and restored at the same locations in another,
108
100
  on top of the implicitly collected `/logs/artifacts/` convention dir. Declare
109
101
  runtime outputs that must cross that boundary. A declared path that is missing at
110
102
  collection time fails the rollout."""
103
+
111
104
  timeout: TaskTimeout = TaskTimeout()
112
105
  resources: TaskResources = TaskResources()
113
106
 
@@ -125,23 +118,10 @@ class WireTaskData(TaskData):
125
118
  model_config = ConfigDict(extra="allow")
126
119
 
127
120
 
128
- # No `default=`: an unparameterized `Trace`'s `task` field must serialize duck-typed
129
- # (a defaulted TypeVar narrows pydantic's serialization to the base `TaskData`, silently
130
- # dropping subclass fields from the wire).
131
121
  DataT = TypeVar("DataT", bound=TaskData)
132
122
  ConfigT = TypeVar("ConfigT", bound=TaskConfig, default=TaskConfig)
133
123
 
134
124
 
135
- def task_data_cls(cls: type) -> type[TaskData]:
136
- """Resolve a task's `TaskData` specialization through its MRO, else `TaskData`."""
137
- return concrete_type(cls, TaskData) or TaskData
138
-
139
-
140
- def task_config_cls(cls: type) -> type[TaskConfig]:
141
- """Resolve a task's `TaskConfig` specialization through its MRO, else `TaskConfig`."""
142
- return concrete_type(cls, TaskConfig) or TaskConfig
143
-
144
-
145
125
  def resolve_server_config(
146
126
  owner: str, config: BaseConfig, server_cls: type, *, sole: bool = True
147
127
  ) -> BaseConfig:
@@ -168,48 +148,20 @@ def resolve_server_config(
168
148
 
169
149
 
170
150
  class Task(Generic[DataT, StateT, ConfigT]):
171
- """Behavior, lifecycle, servers, and scoring for one `TaskData` row.
172
-
173
- Parameterize as `Task[MyData, MyState, MyConfig]`. Construction accepts the row
174
- and an optional config; omitting config builds the declared config type. One task
175
- instance is shared across a rollout group, so per-rollout state belongs on the trace.
176
- """
177
-
178
151
  NEEDS_CONTAINER: ClassVar[bool] = False
152
+ """Whether the task needs a containerized environment (isolated filesystem, ...)."""
179
153
 
180
154
  tools: ClassVar[tuple[type[Toolset], ...]] = ()
181
155
 
182
156
  def __init__(self, data: DataT, config: ConfigT | None = None) -> None:
183
157
  self.data = data
184
- self.config = config if config is not None else task_config_cls(type(self))()
158
+ self.config = config if config is not None else self.config_type()()
185
159
 
186
160
  def with_system_prompt(self, system_prompt: str) -> Self:
187
- """A shallow copy of this task with `data.system_prompt` overridden. Copies the
188
- instance instead of reconstructing via `type(self)(...)`, so a subclass with a
189
- non-`(data, config)` constructor or extra load-time state keeps it. Used to apply the
190
- config-layer / GEPA system prompt (see `TasksetConfig` and `verifiers.v1.gepa`)."""
191
161
  clone = copy.copy(self)
192
162
  clone.data = self.data.model_copy(update={"system_prompt": system_prompt})
193
163
  return clone
194
164
 
195
- def plugged_judges(self) -> list[Judge]:
196
- from verifiers.v1.loaders import load_judge
197
-
198
- return [load_judge(config) for config in self.config.judges]
199
-
200
- def server_config(self, server_cls: type) -> BaseConfig:
201
- """The config a declared server class (`tools`) is built with (see
202
- `resolve_server_config`). Override to pair explicitly."""
203
- return resolve_server_config(
204
- type(self).__name__,
205
- self.config,
206
- server_cls,
207
- sole=len(set(type(self).tools)) == 1,
208
- )
209
-
210
- def tool_servers(self) -> list[Toolset]:
211
- return [cls(self.server_config(cls)) for cls in type(self).tools]
212
-
213
165
  async def setup(self, trace: Trace, runtime: Runtime) -> None:
214
166
  return None
215
167
 
@@ -284,5 +236,29 @@ class Task(Generic[DataT, StateT, ConfigT]):
284
236
  for key, value in items:
285
237
  trace.record_reward(key, value, judge.config.weight)
286
238
 
239
+ @classmethod
240
+ def data_type(cls) -> type[TaskData]:
241
+ return concrete_type(cls, TaskData) or TaskData
242
+
243
+ @classmethod
244
+ def config_type(cls) -> type[TaskConfig]:
245
+ return concrete_type(cls, TaskConfig) or TaskConfig
246
+
247
+ def plugged_judges(self) -> list[Judge]:
248
+ from verifiers.v1.loaders import load_judge
249
+
250
+ return [load_judge(config) for config in self.config.judges]
251
+
252
+ def server_config(self, server_cls: type) -> BaseConfig:
253
+ return resolve_server_config(
254
+ type(self).__name__,
255
+ self.config,
256
+ server_cls,
257
+ sole=len(set(type(self).tools)) == 1,
258
+ )
259
+
260
+ def tool_servers(self) -> list[Toolset]:
261
+ return [cls(self.server_config(cls)) for cls in type(self).tools]
262
+
287
263
 
288
264
  TaskT = TypeVar("TaskT", bound=Task)
verifiers/v1/taskset.py CHANGED
@@ -103,8 +103,6 @@ class Taskset(ABC, Generic[TaskT, TasksetConfigT]):
103
103
  return concrete_type(cls, Task, origin=Taskset) or Task
104
104
 
105
105
  def server_config(self, server_cls: type) -> BaseConfig:
106
- """The config a `tools` entry is built with, resolved off `self.config` (the
107
- taskset config; see `resolve_server_config`). Override to pair explicitly."""
108
106
  return resolve_server_config(
109
107
  type(self).__name__,
110
108
  self.config,
@@ -22,7 +22,7 @@ from collections.abc import Iterator
22
22
  from functools import lru_cache
23
23
  from pathlib import Path
24
24
 
25
- from pydantic import Field
25
+ from pydantic import BaseModel, Field
26
26
 
27
27
  from verifiers.v1.artifacts import Artifact, collect
28
28
  from verifiers.v1.configs.taskset import TasksetConfig
@@ -32,7 +32,6 @@ from verifiers.v1.runtimes import Runtime
32
32
  from verifiers.v1.task import Task, TaskData, TaskResources, TaskTimeout
33
33
  from verifiers.v1.taskset import Taskset
34
34
  from verifiers.v1.trace import Trace
35
- from verifiers.v1.types import StrictBaseModel
36
35
 
37
36
  CACHE = Path.home() / ".cache" / "harbor"
38
37
  HARBOR_INSTALL_HINT = "uv sync --python 3.12 --extra harbor"
@@ -73,12 +72,12 @@ class HarborConfig(TasksetConfig):
73
72
  has what the task needs (e.g. you've pointed the runtime at the right image)."""
74
73
 
75
74
 
76
- class Author(StrictBaseModel):
75
+ class Author(BaseModel):
77
76
  name: str | None = None
78
77
  email: str | None = None
79
78
 
80
79
 
81
- class CollectHook(StrictBaseModel):
80
+ class CollectHook(BaseModel):
82
81
  """One `[[verifier.collect]]` command, run in the agent's box by `finalize`."""
83
82
 
84
83
  command: str
verifiers/v1/trace.py CHANGED
@@ -7,7 +7,7 @@ from collections.abc import Mapping
7
7
  from typing import TYPE_CHECKING, Annotated, Any, Generic, Literal
8
8
 
9
9
  import numpy as np
10
- from pydantic import Field, PrivateAttr
10
+ from pydantic import BaseModel, Field, PrivateAttr
11
11
  from renderers.base import MultiModalData
12
12
  from typing_extensions import TypeVar
13
13
 
@@ -27,7 +27,6 @@ from verifiers.v1.types import (
27
27
  KeptTokens,
28
28
  Messages,
29
29
  Sampling,
30
- StrictBaseModel,
31
30
  Tool,
32
31
  ToolMessage,
33
32
  Usage,
@@ -50,7 +49,7 @@ EXCLUDE_FIELDS: dict = {
50
49
  """Raw tensor fields kept on the msgpack wire but excluded from disk serialization."""
51
50
 
52
51
 
53
- class TimeSpan(StrictBaseModel):
52
+ class TimeSpan(BaseModel):
54
53
  """Wall-clock timestamps with a derived, non-serialized duration in seconds."""
55
54
 
56
55
  start: float = 0.0
@@ -61,34 +60,34 @@ class TimeSpan(StrictBaseModel):
61
60
  return max(0.0, self.end - self.start) if self.end else 0.0
62
61
 
63
62
 
64
- class TimeSplit(StrictBaseModel):
63
+ class TimeSplit(BaseModel):
65
64
  """Records a measured duration in seconds."""
66
65
 
67
66
  duration: float = 0.0
68
67
 
69
68
 
70
- class GenerationSpan(TimeSpan):
69
+ class AgentSpan(TimeSpan):
71
70
  model: TimeSplit = Field(default_factory=TimeSplit)
72
71
  harness: TimeSplit = Field(default_factory=TimeSplit)
73
72
 
74
73
 
75
- class Timing(StrictBaseModel):
74
+ class Timing(BaseModel):
76
75
  start: float = Field(default_factory=time.time)
77
76
  boot: TimeSpan = Field(default_factory=TimeSpan)
78
77
  setup: TimeSpan = Field(default_factory=TimeSpan)
79
- generation: GenerationSpan = Field(default_factory=GenerationSpan)
78
+ agent: AgentSpan = Field(default_factory=AgentSpan)
80
79
  finalize: TimeSpan = Field(default_factory=TimeSpan)
81
80
  scoring: TimeSpan = Field(default_factory=TimeSpan)
82
81
 
83
82
 
84
- class Error(StrictBaseModel):
83
+ class Error(BaseModel):
85
84
  type: str
86
85
  message: str
87
86
  status_code: int | None = None
88
87
  traceback: str | None = None
89
88
 
90
89
 
91
- class VersionInfo(StrictBaseModel):
90
+ class VersionInfo(BaseModel):
92
91
  version: str
93
92
  commit: str | None = None
94
93
 
@@ -104,7 +103,7 @@ def _current_build() -> VersionInfo:
104
103
  AgentConfigT = TypeVar("AgentConfigT", bound=AgentConfig, default=AgentConfig)
105
104
 
106
105
 
107
- class AgentInfo(StrictBaseModel, Generic[AgentConfigT]):
106
+ class AgentInfo(BaseModel, Generic[AgentConfigT]):
108
107
  config: AgentConfigT
109
108
  """The resolved config that rebuilds the agent (`Agent(trace.agent.config)`)."""
110
109
  runtime: RuntimeInfo | None = None
@@ -115,7 +114,7 @@ class AgentInfo(StrictBaseModel, Generic[AgentConfigT]):
115
114
  """Whether this trace's tokens train the run's policy."""
116
115
 
117
116
 
118
- class TraceTask(StrictBaseModel, Generic[DataT]):
117
+ class TraceTask(BaseModel, Generic[DataT]):
119
118
  """The task as recorded on the trace, self-describing without the run's config."""
120
119
 
121
120
  type: str
@@ -124,7 +123,7 @@ class TraceTask(StrictBaseModel, Generic[DataT]):
124
123
  """The (immutable) row being solved."""
125
124
 
126
125
 
127
- class Reward(StrictBaseModel):
126
+ class Reward(BaseModel):
128
127
  score: float
129
128
  weight: float = 1.0
130
129
 
@@ -133,14 +132,14 @@ class Reward(StrictBaseModel):
133
132
  return self.score * self.weight
134
133
 
135
134
 
136
- class EvalRunInfo(StrictBaseModel):
135
+ class EvalRunInfo(BaseModel):
137
136
  type: Literal["eval"] = "eval"
138
137
 
139
138
  id: str
140
139
  step: int | None = None
141
140
 
142
141
 
143
- class TrainRunInfo(StrictBaseModel):
142
+ class TrainRunInfo(BaseModel):
144
143
  type: Literal["train"] = "train"
145
144
 
146
145
  id: str
@@ -151,7 +150,7 @@ RunInfo = Annotated[EvalRunInfo | TrainRunInfo, Field(discriminator="type")]
151
150
  """The run a trace belongs to, discriminated on `type`."""
152
151
 
153
152
 
154
- class ModelCall(StrictBaseModel):
153
+ class ModelCall(BaseModel):
155
154
  """A model call, automatically recorded at intercept time."""
156
155
 
157
156
  node: int | None = None
@@ -172,7 +171,7 @@ class ModelCall(StrictBaseModel):
172
171
  """The failure that ended this call, coupled to the exchange that caused it."""
173
172
 
174
173
 
175
- class Branch(StrictBaseModel):
174
+ class Branch(BaseModel):
176
175
  """A root-to-leaf graph path; each branch becomes one training sample."""
177
176
 
178
177
  index: int
@@ -298,7 +297,7 @@ class Branch(StrictBaseModel):
298
297
  return self.num_total_tokens - self.num_output_tokens
299
298
 
300
299
 
301
- class Trace(StrictBaseModel, Generic[DataT, StateT, AgentConfigT]):
300
+ class Trace(BaseModel, Generic[DataT, StateT, AgentConfigT]):
302
301
  version: int = TRACE_VERSION
303
302
  """The trace schema this trace serializes as."""
304
303
  id: str = Field(default_factory=lambda: uuid.uuid4().hex)
@@ -492,14 +491,14 @@ class Trace(StrictBaseModel, Generic[DataT, StateT, AgentConfigT]):
492
491
  if self.stop_condition is None:
493
492
  self.stop_condition = condition
494
493
 
495
- def split_generation(self) -> None:
496
- """Split the generation span into model and harness time."""
497
- gen = self.timing.generation
498
- if not gen.end:
494
+ def split_agent_time(self) -> None:
495
+ """Split the agent span into model and harness time."""
496
+ span = self.timing.agent
497
+ if not span.end:
499
498
  return
500
499
  model = sum(call.time.duration for call in self.calls)
501
- gen.model.duration = min(model, gen.duration)
502
- gen.harness.duration = gen.duration - gen.model.duration
500
+ span.model.duration = min(model, span.duration)
501
+ span.harness.duration = span.duration - span.model.duration
503
502
 
504
503
  def record_error(self, error: Exception) -> None:
505
504
  """Record an error, and stop the trace as failed."""
verifiers/v1/types.py CHANGED
@@ -7,20 +7,16 @@ from renderers.base import MultiModalData
7
7
  from typing_extensions import TypedDict
8
8
 
9
9
 
10
- class StrictBaseModel(BaseModel):
11
- model_config = ConfigDict(extra="forbid")
12
-
13
-
14
- class TextContentPart(StrictBaseModel):
10
+ class TextContentPart(BaseModel):
15
11
  type: Literal["text"] = "text"
16
12
  text: str
17
13
 
18
14
 
19
- class ImageUrlSource(StrictBaseModel):
15
+ class ImageUrlSource(BaseModel):
20
16
  url: str
21
17
 
22
18
 
23
- class ImageUrlContentPart(StrictBaseModel):
19
+ class ImageUrlContentPart(BaseModel):
24
20
  type: Literal["image_url"] = "image_url"
25
21
  image_url: ImageUrlSource
26
22
 
@@ -57,24 +53,24 @@ def content_text(content: "MessageContent | None") -> str:
57
53
  )
58
54
 
59
55
 
60
- class SystemMessage(StrictBaseModel):
56
+ class SystemMessage(BaseModel):
61
57
  role: Literal["system"] = "system"
62
58
  content: MessageContent
63
59
 
64
60
 
65
- class UserMessage(StrictBaseModel):
61
+ class UserMessage(BaseModel):
66
62
  role: Literal["user"] = "user"
67
63
  content: MessageContent
68
64
 
69
65
 
70
- class ToolCall(StrictBaseModel):
66
+ class ToolCall(BaseModel):
71
67
  id: str
72
68
  name: str
73
69
  arguments: str
74
70
  """Raw JSON string of arguments, exactly as the model emitted it."""
75
71
 
76
72
 
77
- class AssistantMessage(StrictBaseModel):
73
+ class AssistantMessage(BaseModel):
78
74
  role: Literal["assistant"] = "assistant"
79
75
  content: str | None = None
80
76
  reasoning_content: str | None = None
@@ -83,7 +79,7 @@ class AssistantMessage(StrictBaseModel):
83
79
  """Opaque native items replayed to preserve signed or encrypted reasoning state."""
84
80
 
85
81
 
86
- class ToolMessage(StrictBaseModel):
82
+ class ToolMessage(BaseModel):
87
83
  role: Literal["tool"] = "tool"
88
84
  tool_call_id: str
89
85
  content: MessageContent
@@ -98,7 +94,7 @@ Message = Annotated[
98
94
  Messages = list[Message]
99
95
 
100
96
 
101
- class Tool(StrictBaseModel):
97
+ class Tool(BaseModel):
102
98
  name: str
103
99
  description: str
104
100
  parameters: dict[str, Any]
@@ -108,7 +104,7 @@ class Tool(StrictBaseModel):
108
104
  FinishReason = Literal["stop", "length", "tool_calls"] | None
109
105
 
110
106
 
111
- class Usage(StrictBaseModel):
107
+ class Usage(BaseModel):
112
108
  """Provider token accounting.
113
109
 
114
110
  `prompt_tokens` excludes cache reads; `input_tokens` adds them back. Reasoning tokens
@@ -191,10 +187,10 @@ class KeptTokens:
191
187
  counts: Any
192
188
 
193
189
 
194
- class TurnTokens(StrictBaseModel):
190
+ class TurnTokens(BaseModel):
195
191
  """Training tokens from renderer tokenization or provider-returned token IDs."""
196
192
 
197
- model_config = ConfigDict(extra="forbid", arbitrary_types_allowed=True)
193
+ model_config = ConfigDict(arbitrary_types_allowed=True)
198
194
 
199
195
  prompt_ids: list[int] = Field(default_factory=list)
200
196
  completion_ids: list[int] = Field(default_factory=list)
@@ -220,7 +216,7 @@ class TurnTokens(StrictBaseModel):
220
216
  kept_tokens: KeptTokens | None = Field(default=None, exclude=True)
221
217
 
222
218
 
223
- class Response(StrictBaseModel):
219
+ class Response(BaseModel):
224
220
  id: str
225
221
  created: int
226
222
  model: str
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: verifiers
3
- Version: 0.2.2.dev53
3
+ Version: 0.2.2.dev55
4
4
  Summary: Verifiers: Environments for LLM Reinforcement Learning
5
5
  Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
6
6
  Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
@@ -167,42 +167,42 @@ verifiers/utils/threaded_sandbox_client.py,sha256=XRzYhWBpMev7kw_aNMWAy6lfa-9uAn
167
167
  verifiers/utils/tool_utils.py,sha256=gInWZODWQUZN4TyEMuuITyOUm9qlHJxtcvr1zZl_zJs,1020
168
168
  verifiers/utils/usage_utils.py,sha256=Mw6WC78r6EsLyRFdP9ICveqptkjizbSkAC86ZH0geVU,3918
169
169
  verifiers/utils/version_utils.py,sha256=39gV8-mNF9GspAgKloLo8oy7_MVluMrwGPrntQv2hx4,2656
170
- verifiers/v1/__init__.py,sha256=uqq1Jzek51cYa2TyHqzDwU6zlOmD3wKLF063Kjh_iN0,7521
170
+ verifiers/v1/__init__.py,sha256=Im5qTZy9mH6PRTUi-1V6W7EWKZzJyTWLGKMYALbWz_w,7467
171
171
  verifiers/v1/agent.py,sha256=dj4GabQSs6Wzwbph5Rlx-efsVQsSuWV2YI5hfvq5mGA,31990
172
- verifiers/v1/artifacts.py,sha256=G-YTfTAYA35zxKzM-jK60TprndfQevPA65hWDw6ae7Y,5877
172
+ verifiers/v1/artifacts.py,sha256=5eKOjDSUWwj7GrhxILMHdz2ngEONWMdcTGlBcuGgrLU,5834
173
173
  verifiers/v1/decorators.py,sha256=9CcCwgZlRjVGPqT5G1CFJUHP4OIMuBGkUJZkcI6dnbk,3865
174
- verifiers/v1/env.py,sha256=m-1TNzs9YyFbMkoW6c54vI5xG4jnW3LvLgvJhOgbg04,18447
175
- verifiers/v1/episode.py,sha256=KWyM9ovTphEIo2260BYu2HvtnOOTxjLRDPrF_jTCiuA,2497
174
+ verifiers/v1/env.py,sha256=a7cKeecsm4d1lZPhRJy-m7FFWHrHF2msOdqma7o5I6M,18468
175
+ verifiers/v1/episode.py,sha256=YtuRTV_PpwDglXFeN6UNg8IswpP1HseFLCYtascGMNI,3218
176
176
  verifiers/v1/errors.py,sha256=slZnrtqMo_BgXQV46wJJhV6vvAAnVsuCbVGp2vBVWfs,6888
177
- verifiers/v1/graph.py,sha256=-j8MNdIv1r5ifajv5ggnGPsfBNB74IO3WKLcrpYh9hg,29102
177
+ verifiers/v1/graph.py,sha256=aBAzh1Ibp3klKNYjvzNyJSbjV8Oi5TJ791dN7o6ex1E,29070
178
178
  verifiers/v1/harness.py,sha256=cDiDuhp29aqL4CxiqSZz-IT0QlOcVlye_vTTHZblWo8,10948
179
- verifiers/v1/judge.py,sha256=duZcWYTynzqShs8vfbExpFlgLS-JA9NFqdmCRLg3sbc,9393
180
- verifiers/v1/legacy.py,sha256=nkbdhFJZN1QzsjE42AB_A_c1zrk3_Y8zR07srZJS2ck,22584
179
+ verifiers/v1/judge.py,sha256=hISmuXidKCMROni9_4q4I-EeoO_uEieafGd2WBjRNKM,9338
180
+ verifiers/v1/legacy.py,sha256=ISk3Bix9Hy5EjZtyV9QO0LNyDNDklBHVjFokOrw_3U8,22628
181
181
  verifiers/v1/loaders.py,sha256=NSIBdNU3JVeyddL-dM4dJt6khCHyjCgzJuFtLrgiwD0,9989
182
182
  verifiers/v1/push.py,sha256=MWZmvyj0iyzNIO0xRhlkCXtFF1plq6EJi8rWlvrFtvI,11363
183
183
  verifiers/v1/retries.py,sha256=quUGVruUkZx-9WQcFiocMfL_M056voU8gwDqsuLR3f4,5412
184
- verifiers/v1/rollout.py,sha256=GJzHgJFlw8O1vDtWM__2JhSiG2GqbfjtH2ri1MRfw3g,20495
184
+ verifiers/v1/rollout.py,sha256=pmF4sgyHDRUUTnOxOWgBsPP8x6L_KFGaeR_dDKziANs,20470
185
185
  verifiers/v1/scoring.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
186
186
  verifiers/v1/session.py,sha256=J6ZPuvsg6vMTJAs9dMUMtfdjCuvp6Hom877DVdKFKlY,6912
187
- verifiers/v1/state.py,sha256=mcQBl9ItRPVrcVr7YzHLbEqKdDSfybhtWD8XDEkeui0,717
188
- verifiers/v1/task.py,sha256=Ofd3Sh8bfFeEy6nhgeHV0CT2-9lAoGypQyVYhtSP1oc,12206
189
- verifiers/v1/taskset.py,sha256=RPFkpRkeEyPX557_mK-JwraBuUKsguiCUKSFyotTclU,4636
190
- verifiers/v1/trace.py,sha256=9KqBeGaRUtWh6Gpdf6G1c7UZso0DB9Mjmtkzl6I_oYE,19458
191
- verifiers/v1/types.py,sha256=5tZyG4r4bLJ9a14oybHy7bSCpt5X5A17eRwmIszuRo8,9118
187
+ verifiers/v1/state.py,sha256=pcpN2V6tX7rIO5XdtdUYay1j0ktEwrMtq7w1ApfsPdE,675
188
+ verifiers/v1/task.py,sha256=pM2S4YU56jTuvvwHe00r1pBupQ42pvMUzpQhUXqpAPk,10232
189
+ verifiers/v1/taskset.py,sha256=5ffTlnmiN2GDePhonGSJ1a7iTseLPOC0DkwaVwf7uY4,4465
190
+ verifiers/v1/trace.py,sha256=_bGHOrE6fRLGm0JYc0PrrrO3lyr-qqOF5Y1bHfViwR0,19347
191
+ verifiers/v1/types.py,sha256=1PxamJspmoTc3OlFZAwH6o_2i_Tg9ZNvHxttIHWIxq8,8948
192
192
  verifiers/v1/acp/__init__.py,sha256=9dwH6fLopNndRmcpRSYC8ghzzMquaLfgh645QM-Dwic,2189
193
193
  verifiers/v1/acp/_runner.py,sha256=FqEnHR0Q_YrCt_EuaJqiqTpd2_kFa41FqcEunwOhvTI,6856
194
194
  verifiers/v1/cli/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
195
- verifiers/v1/cli/debug.py,sha256=mRBPlYjc93o-4sooVSJXl_H3zALC2vhAl3EBzN-DpoM,11507
195
+ verifiers/v1/cli/debug.py,sha256=-jeEV_Q5tV8SOkWc1GdPZcXZWON_jU5gjeO4YoZm79o,11482
196
196
  verifiers/v1/cli/gepa.py,sha256=hLIbS_4jBx0vvy4PXoePR5riFXcNvzQCMixZ-g2OR_U,4026
197
197
  verifiers/v1/cli/init.py,sha256=dVpX1YMz7DFmHgskAsPjP-HmVTGoZfC0ZUyE-T4I6l0,8322
198
198
  verifiers/v1/cli/output.py,sha256=pMS5Z0EBgyAVShNCAk9jVURNeLOaTfeRWpt0J8hHjq8,5890
199
- verifiers/v1/cli/replay.py,sha256=Jjyi61fl20xjGArdXTwZ3qKoSkC0iQ6WLSrvKkURdeI,10179
199
+ verifiers/v1/cli/replay.py,sha256=DAh9kB5PB-ppZLGhxdJwrCPch_Yky5HKhplwGFhKEz0,10164
200
200
  verifiers/v1/cli/resolve.py,sha256=OsNw4C3r9xOVFa-mminJFcex_bAsJ6EQ8K7A_H8u2kQ,4370
201
201
  verifiers/v1/cli/serve.py,sha256=_mdJaHVVe4pXzcpAot0ugcmiN3ra2gS4GrGmLghnX5M,2866
202
202
  verifiers/v1/cli/validate.py,sha256=lZ_wWnLgIWH5rUlMtwzvF3uS_cFZA4ZqBowpzdI4gos,10260
203
203
  verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
204
204
  verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
205
- verifiers/v1/cli/dashboard/eval.py,sha256=i_Qn9enCU7aas-VTVAJocPNbTuHSUHThwQnls-3_kc4,34265
205
+ verifiers/v1/cli/dashboard/eval.py,sha256=QJlBKRvEaxBYWZ-WQQfJqaBZoVDIO3MnJnzeD3VusEI,33436
206
206
  verifiers/v1/cli/dashboard/replay.py,sha256=CvRVaf0dUum5v0IzPQbFPPikBQmTrkMeaunGfVMTWHc,2755
207
207
  verifiers/v1/cli/dashboard/validate.py,sha256=rLsQ_31DjTIZpC_VO7zSG41ncUEBxqCx8mL9nTnGiMc,3650
208
208
  verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
@@ -239,7 +239,7 @@ verifiers/v1/dialects/chat.py,sha256=R_otqi5N0snB1QVKIYloNeLHR9kc7OmI06HYg1ir_Lw
239
239
  verifiers/v1/dialects/responses.py,sha256=ismnxUikPcPUIm80FUTPBbj-Cn1e0yGAsISMEseAWRc,13679
240
240
  verifiers/v1/envs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
241
241
  verifiers/v1/envs/agentic_judge/__init__.py,sha256=6vwtMCQ_cuvuubCF5-nrE5W7gGgzE-LIPKorkM5DXdw,309
242
- verifiers/v1/envs/agentic_judge/env.py,sha256=NEv1AUbOrGSFDJNzMVmAUkNVk89L1Cpyu36Ndrg3xFI,17003
242
+ verifiers/v1/envs/agentic_judge/env.py,sha256=QRYxs8M5ivk7jnEwPgYfCLVZTBQOuSc64xydoBixXDs,16961
243
243
  verifiers/v1/envs/best_of_n/__init__.py,sha256=mubASiwXMNIhnQFDccNAc7yZ0OFsOVK8hH8WIx9b2A4,119
244
244
  verifiers/v1/envs/best_of_n/env.py,sha256=KEUJBCVqblnXP_lEwm3mjGdaikBMkecz8UjOFPBhxc0,1987
245
245
  verifiers/v1/envs/single_agent/__init__.py,sha256=r0edAc_6gtHBhPzeBGVcdHZZqSdCPZnLMje6eVgyz5U,138
@@ -288,7 +288,7 @@ verifiers/v1/interception/tunnel/prime.py,sha256=K2n2daP_iNoO-087owAYHBkR7z0PBNn
288
288
  verifiers/v1/judges/__init__.py,sha256=MUIBWcx6c70BykrTDHhB0ITIyJPxXe6xDBou4VM7jDQ,286
289
289
  verifiers/v1/judges/reference.py,sha256=iEVHw-iJ6dbwLOOqrvWrEMJOcUEYJ0YspT04rZz7Sd4,3868
290
290
  verifiers/v1/judges/reference.txt,sha256=Ej35kGXiT2uJ0LcHIeez49rCS_9uHmCQKoVdn93V6A8,353
291
- verifiers/v1/judges/rubric.py,sha256=kB0yWEtfC60zJBCVjOKBO00YtH8-AOofzPkuyccVTtQ,11940
291
+ verifiers/v1/judges/rubric.py,sha256=PglrJhMQR6scFyfz20jVlLQEzeqXKV2P6r0K0K7jjBE,11916
292
292
  verifiers/v1/judges/rubric.txt,sha256=3KA2ZeYYdcDweJhUGopxu9EFgARhu2aWft5S5cj8Bc8,323
293
293
  verifiers/v1/mcp/__init__.py,sha256=pm8ZWBiL1q1tY4wHvJzM_QrZ9oR8aGm-bIRWiRJsgHI,408
294
294
  verifiers/v1/mcp/launch.py,sha256=lRtcl98BO7imC44bsxO2yljwRQR_KCvJ6P-bd_hS5L8,18444
@@ -305,11 +305,11 @@ verifiers/v1/runtimes/docker/egress.py,sha256=37UXgNX9gnoLxjMg4KZaJIGoApCAM6MK22
305
305
  verifiers/v1/serve/__init__.py,sha256=sr_zkDzAHnOcQm-QRm8xwJBmCMFD-xhKIFPaDZvMbUs,641
306
306
  verifiers/v1/serve/client.py,sha256=amUhf4cFJ1xCb8kw_AoVviwN543Wknx8LzImq1ZpaiQ,7057
307
307
  verifiers/v1/serve/pool.py,sha256=e9Nc3nQKEyV9YqyKHl_RuqQ84zWVa8z2J1rc3rfCadM,14828
308
- verifiers/v1/serve/server.py,sha256=VBOD3Gzq4p0ZD_8xgtG039i3JaefhOH0MmR6cxG56iY,9705
308
+ verifiers/v1/serve/server.py,sha256=IfJunpYX3MmNDToEcY-QHxDGfbZnK2i3C1-sgft6xfw,9687
309
309
  verifiers/v1/serve/types.py,sha256=y8Cf9wKOZRKDXcThFWVygTFHITJofOinGiIzwjQGCUE,2674
310
310
  verifiers/v1/tasksets/__init__.py,sha256=b62O4WDNbGfJH_4NjYmznn5otWovdfvUgJZyoIZyLh8,596
311
311
  verifiers/v1/tasksets/harbor/__init__.py,sha256=wwVGA8N7E2Q2qA5WGHuLRrn4DgG5oaRHrCS2RW3NXVY,195
312
- verifiers/v1/tasksets/harbor/taskset.py,sha256=pkXYS1vknI66SXQV86ypUEZ4kCBdVH2QjMuGjWt8UC0,18283
312
+ verifiers/v1/tasksets/harbor/taskset.py,sha256=w7o7yGM-FmvLEW4zNvwutvOX4n9cq8eQWwb7WxUYtSA,18235
313
313
  verifiers/v1/tasksets/lean/__init__.py,sha256=oyv-GbJQBvCyGwo7Ysxm2NjZW-aX4j1uDoHU0f-E7xQ,767
314
314
  verifiers/v1/tasksets/lean/scoring.py,sha256=sfzsT6MUK0zdMgAQ57_P_0CBaPHy3M7go71QMr5hDrs,10599
315
315
  verifiers/v1/tasksets/lean/taskset.py,sha256=Fcn4UgTcdjMYnJ4CThaFZ28Rz_QDx0e21vJdW_vp12g,9019
@@ -330,8 +330,8 @@ verifiers/v1/utils/logging.py,sha256=OcMHA6NsYux3oIzjPuI95rDWmFBHNcHDjZeNIhXTX-Y
330
330
  verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
331
331
  verifiers/v1/utils/sampling.py,sha256=52TJ5-Hmpvp4YSgKNR-LmAPVwm4RDcH-o2-BnbVTmYQ,965
332
332
  verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
333
- verifiers-0.2.2.dev53.dist-info/METADATA,sha256=vvfSCIqaWIr6egUCiFZ505i0TPbN5Cc0boYECWJ-sMg,4545
334
- verifiers-0.2.2.dev53.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
335
- verifiers-0.2.2.dev53.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
336
- verifiers-0.2.2.dev53.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
337
- verifiers-0.2.2.dev53.dist-info/RECORD,,
333
+ verifiers-0.2.2.dev55.dist-info/METADATA,sha256=Jcl9dMFMROAPjPXzDCMNCz_r43jDq4Xp2QjEZZ0Z9b4,4545
334
+ verifiers-0.2.2.dev55.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
335
+ verifiers-0.2.2.dev55.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
336
+ verifiers-0.2.2.dev55.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
337
+ verifiers-0.2.2.dev55.dist-info/RECORD,,