verifiers 0.2.2.dev23__py3-none-any.whl → 0.2.2.dev25__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. verifiers/v1/__init__.py +14 -33
  2. verifiers/v1/agent.py +34 -73
  3. verifiers/v1/cli/dashboard/eval.py +19 -11
  4. verifiers/v1/cli/dashboard/validate.py +1 -1
  5. verifiers/v1/cli/debug.py +11 -3
  6. verifiers/v1/cli/eval/main.py +1 -1
  7. verifiers/v1/cli/eval/resume.py +1 -1
  8. verifiers/v1/cli/eval/runner.py +2 -2
  9. verifiers/v1/cli/init.py +1 -1
  10. verifiers/v1/cli/output.py +1 -1
  11. verifiers/v1/cli/replay.py +5 -2
  12. verifiers/v1/cli/serve.py +2 -2
  13. verifiers/v1/cli/validate.py +1 -1
  14. verifiers/v1/configs/__init__.py +4 -7
  15. verifiers/v1/configs/agent.py +77 -0
  16. verifiers/v1/configs/cli/__init__.py +9 -0
  17. verifiers/v1/configs/{debug.py → cli/debug.py} +2 -2
  18. verifiers/v1/configs/cli/env.py +167 -0
  19. verifiers/v1/configs/{eval.py → cli/eval.py} +1 -1
  20. verifiers/v1/configs/{replay.py → cli/replay.py} +1 -1
  21. verifiers/v1/configs/{serve.py → cli/serve.py} +1 -1
  22. verifiers/v1/configs/{validate.py → cli/validate.py} +1 -1
  23. verifiers/v1/configs/env.py +139 -150
  24. verifiers/v1/configs/harness.py +42 -0
  25. verifiers/v1/configs/judge.py +73 -0
  26. verifiers/v1/configs/retries.py +17 -0
  27. verifiers/v1/configs/task.py +32 -0
  28. verifiers/v1/configs/taskset.py +20 -0
  29. verifiers/v1/env.py +12 -155
  30. verifiers/v1/envs/agentic_judge/env.py +5 -5
  31. verifiers/v1/envs/single_agent/env.py +4 -2
  32. verifiers/v1/episode.py +7 -5
  33. verifiers/v1/gepa/config.py +2 -2
  34. verifiers/v1/gepa/reflection.py +2 -1
  35. verifiers/v1/harness.py +5 -39
  36. verifiers/v1/harnesses/bash/harness.py +2 -1
  37. verifiers/v1/harnesses/claude_code/harness.py +2 -1
  38. verifiers/v1/harnesses/codex/harness.py +2 -1
  39. verifiers/v1/harnesses/kimi_code/harness.py +2 -1
  40. verifiers/v1/harnesses/mini_swe_agent/harness.py +2 -1
  41. verifiers/v1/harnesses/null/harness.py +2 -1
  42. verifiers/v1/harnesses/pi/harness.py +2 -1
  43. verifiers/v1/harnesses/pool/harness.py +2 -1
  44. verifiers/v1/harnesses/rlm/harness.py +2 -1
  45. verifiers/v1/harnesses/terminus_2/harness.py +2 -1
  46. verifiers/v1/judge.py +7 -66
  47. verifiers/v1/judges/reference.py +1 -1
  48. verifiers/v1/judges/rubric.py +2 -7
  49. verifiers/v1/loaders.py +8 -4
  50. verifiers/v1/mcp/launch.py +7 -7
  51. verifiers/v1/push.py +1 -1
  52. verifiers/v1/retries.py +1 -15
  53. verifiers/v1/rollout.py +7 -7
  54. verifiers/v1/runtimes/__init__.py +2 -0
  55. verifiers/v1/runtimes/base.py +43 -3
  56. verifiers/v1/runtimes/docker/__init__.py +8 -35
  57. verifiers/v1/runtimes/prime.py +57 -4
  58. verifiers/v1/serve/client.py +3 -4
  59. verifiers/v1/serve/pool.py +1 -1
  60. verifiers/v1/serve/server.py +1 -1
  61. verifiers/v1/serve/types.py +4 -6
  62. verifiers/v1/task.py +5 -29
  63. verifiers/v1/taskset.py +2 -16
  64. verifiers/v1/tasksets/harbor/taskset.py +2 -1
  65. verifiers/v1/tasksets/lean/taskset.py +4 -2
  66. verifiers/v1/trace.py +34 -21
  67. verifiers/v1/utils/compile.py +8 -14
  68. {verifiers-0.2.2.dev23.dist-info → verifiers-0.2.2.dev25.dist-info}/METADATA +2 -2
  69. {verifiers-0.2.2.dev23.dist-info → verifiers-0.2.2.dev25.dist-info}/RECORD +73 -65
  70. /verifiers/v1/configs/{init.py → cli/init.py} +0 -0
  71. {verifiers-0.2.2.dev23.dist-info → verifiers-0.2.2.dev25.dist-info}/WHEEL +0 -0
  72. {verifiers-0.2.2.dev23.dist-info → verifiers-0.2.2.dev25.dist-info}/entry_points.txt +0 -0
  73. {verifiers-0.2.2.dev23.dist-info → verifiers-0.2.2.dev25.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/__init__.py CHANGED
@@ -13,25 +13,16 @@ from verifiers.v1.clients import (
13
13
  resolve_client,
14
14
  )
15
15
  from verifiers.v1.decorators import metric, reward, stop, tool
16
- from verifiers.v1.agent import (
17
- Agent,
18
- AgentConfig,
19
- Agents,
20
- Interaction,
21
- Segment,
22
- make_agent,
23
- )
24
- from verifiers.v1.configs.env import (
16
+ from verifiers.v1.configs.agent import AgentConfig
17
+ from verifiers.v1.agent import Agent, Agents, Interaction, Segment, make_agent
18
+ from verifiers.v1.configs.cli.env import (
25
19
  ElasticPoolConfig,
26
20
  EnvServerConfig,
27
21
  StaticPoolConfig,
28
22
  pool_serve_kwargs,
29
23
  )
30
- from verifiers.v1.env import (
31
- EnvConfig,
32
- Env,
33
- default_agent_harness,
34
- )
24
+ from verifiers.v1.configs.env import EnvConfig, default_agent_harness
25
+ from verifiers.v1.env import Env
35
26
  from verifiers.v1.envs.single_agent import SingleAgentEnv, SingleAgentEnvConfig
36
27
  from verifiers.v1.errors import (
37
28
  EnvError,
@@ -44,15 +35,10 @@ from verifiers.v1.errors import (
44
35
  ToolsetError,
45
36
  TunnelError,
46
37
  )
47
- from verifiers.v1.harness import Harness, HarnessConfig
48
- from verifiers.v1.judge import (
49
- Judge,
50
- JudgeConfig,
51
- JudgeResponse,
52
- Judges,
53
- JudgeSamplingConfig,
54
- JudgeView,
55
- )
38
+ from verifiers.v1.configs.harness import HarnessConfig
39
+ from verifiers.v1.harness import Harness
40
+ from verifiers.v1.configs.judge import JudgeConfig, JudgeSamplingConfig, Judges
41
+ from verifiers.v1.judge import Judge, JudgeResponse, JudgeView
56
42
  from verifiers.v1.judges import (
57
43
  ReferenceJudge,
58
44
  ReferenceJudgeConfig,
@@ -86,7 +72,7 @@ from verifiers.v1.scoring import (
86
72
  read_answer_file_or_last_reply as read_answer_file_or_last_reply,
87
73
  verify_boxed_math_answer as verify_boxed_math_answer,
88
74
  )
89
- from verifiers.v1.retries import RetryConfig
75
+ from verifiers.v1.configs.retries import RetryConfig
90
76
  from verifiers.v1.utils.git import (
91
77
  PATCH_CAP_BYTES as PATCH_CAP_BYTES,
92
78
  capture_patch as capture_patch,
@@ -102,15 +88,10 @@ from verifiers.v1.runtimes import (
102
88
  SubprocessConfig,
103
89
  )
104
90
  from verifiers.v1.state import State, StateT
105
- from verifiers.v1.task import (
106
- Task,
107
- TaskConfig,
108
- TaskData,
109
- TaskResources,
110
- TaskTimeout,
111
- WireTaskData,
112
- )
113
- from verifiers.v1.taskset import Taskset, TasksetConfig
91
+ from verifiers.v1.configs.task import TaskConfig
92
+ from verifiers.v1.task import Task, TaskData, TaskResources, TaskTimeout, WireTaskData
93
+ from verifiers.v1.configs.taskset import TasksetConfig
94
+ from verifiers.v1.taskset import Taskset
114
95
  from verifiers.v1.mcp import (
115
96
  Toolset,
116
97
  SharedToolsetConfig,
verifiers/v1/agent.py CHANGED
@@ -14,23 +14,21 @@ from contextlib import asynccontextmanager, nullcontext
14
14
  from dataclasses import dataclass
15
15
  from typing import AsyncIterator
16
16
 
17
- from pydantic import SerializeAsAny, model_validator
18
- from pydantic_config import BaseConfig
19
17
 
18
+ from verifiers.v1.configs.agent import AgentConfig, TimeoutConfig
20
19
  from verifiers.v1.clients import (
21
20
  Client,
22
- ClientConfig,
23
21
  EvalClientConfig,
24
22
  ModelContext,
25
23
  resolve_client,
26
24
  )
27
- from verifiers.v1.harness import Harness, HarnessConfig
25
+ from verifiers.v1.harness import Harness
28
26
  from verifiers.v1.interception import Interception, InterceptionServer
29
27
  from verifiers.v1.mcp import SharedToolServer
30
- from verifiers.v1.retries import RetryConfig, backoff, trace_should_retry
28
+ from verifiers.v1.retries import backoff, trace_should_retry
31
29
  from verifiers.v1.rollout import RolloutRun, _as_messages
32
30
  from verifiers.v1.runtimes import (
33
- DockerConfig,
31
+ NetworkPolicyConfig,
34
32
  Runtime,
35
33
  RuntimeConfig,
36
34
  SubprocessConfig,
@@ -44,7 +42,6 @@ from verifiers.v1.types import (
44
42
  AssistantMessage,
45
43
  Messages,
46
44
  Sampling,
47
- SamplingConfig,
48
45
  ToolMessage,
49
46
  UserMessage,
50
47
  )
@@ -54,57 +51,9 @@ from verifiers.v1.utils.compile import (
54
51
  validate_pairing,
55
52
  )
56
53
 
57
- logger = logging.getLogger(__name__)
58
-
54
+ __all__ = ["Agent", "AgentConfig", "Agents", "TimeoutConfig", "make_agent"]
59
55
 
60
- class TimeoutConfig(BaseConfig):
61
- """Per-agent wall-clock timeouts per rollout stage, in seconds (None = no
62
- limit); each stage falls back to the task's own `TaskTimeout` when unset. An
63
- interaction's rollout budget is cumulative across its active harness segments
64
- and pauses while the caller computes the next user turn."""
65
-
66
- setup: float | None = None # one shared budget: task setup + provisioning
67
- rollout: float | None = None
68
- finalize: float | None = None
69
- scoring: float | None = None
70
-
71
-
72
- class AgentConfig(BaseConfig):
73
- """One env agent: who plays it, and its per-run caps. It pins only what
74
- makes it a different actor; everything unpinned falls back — the model context
75
- to the run's own, the harness to the taskset's default."""
76
-
77
- harness: SerializeAsAny[HarnessConfig] | None = None
78
- """The agent's program + runtime policy (None = the taskset's default harness)."""
79
- model: str | None = None
80
- """Model id (None = the run's model, i.e. the policy under evaluation/training)."""
81
- client: ClientConfig | None = None
82
- """Endpoint override (None = the run's client)."""
83
- sampling: SamplingConfig | None = None
84
- """Sampling override (None = the run's sampling)."""
85
- timeout: TimeoutConfig = TimeoutConfig()
86
- retries: RetryConfig = RetryConfig()
87
- """Whole-run retries: rerun this agent's rollout while its trace ends with a
88
- retryable error (never into a borrowed box)."""
89
- max_turns: int | None = None
90
- """Max model turns per run (None = no limit). Framework-enforced (the
91
- interception server refuses turns past it), so it applies to any harness."""
92
- max_input_tokens: int | None = None
93
- max_output_tokens: int | None = None
94
- max_total_tokens: int | None = None
95
- """Token caps per run (None = no limit); framework-enforced between turns."""
96
-
97
- @model_validator(mode="before")
98
- @classmethod
99
- def _resolve_harness(cls, data):
100
- """Narrow a pinned `harness` to its concrete config type by `id` (absent
101
- stays None = the taskset's default). The lazy import keeps class-body
102
- `AgentConfig()` defaults constructible while this module initializes."""
103
- if isinstance(data, dict) and data.get("harness") is not None:
104
- from verifiers.v1.loaders import harness_config_type, narrow_plugin_field
105
-
106
- narrow_plugin_field(data, "harness", harness_config_type, "bash")
107
- return data
56
+ logger = logging.getLogger(__name__)
108
57
 
109
58
 
110
59
  def _check_borrowed_placement(
@@ -114,20 +63,26 @@ def _check_borrowed_placement(
114
63
  be honored. Reject requirements that cannot be applied to the running box; an
115
64
  image mismatch on a container only warns, since sharing its world is the point."""
116
65
  task_policy = "*" not in task.data.network_allow or bool(task.data.network_block)
117
- base_docker = base_config if isinstance(base_config, DockerConfig) else None
118
- if task_policy or (base_docker is not None and base_docker.network_isolated):
66
+ base_policy = base_config if isinstance(base_config, NetworkPolicyConfig) else None
67
+ if task_policy or (base_policy is not None and base_policy.network_restricted):
119
68
  config = runtime.config
120
- if not isinstance(config, DockerConfig):
69
+ if not isinstance(config, NetworkPolicyConfig):
121
70
  raise ValueError(
122
- f"task {task.data.idx!r} requires a Docker URL network policy, but "
123
- f"borrowed runtime {runtime.name!r} is not Docker-backed; use "
71
+ f"task {task.data.idx!r} requires a framework-aware network policy, "
72
+ f"but borrowed runtime {runtime.name!r} does not support one; use "
124
73
  "agent.provision(task)"
125
74
  )
126
- policy_base = (
127
- base_docker if base_docker is not None else DockerConfig(allow=["*"])
75
+ if base_policy is not None and type(config) is not type(base_policy):
76
+ raise ValueError(
77
+ f"the configured {base_policy.type} network policy cannot be applied "
78
+ f"to borrowed {config.type} runtime {runtime.name!r}; use "
79
+ "agent.provision(task)"
80
+ )
81
+ policy_base = base_policy or config.model_copy(
82
+ update={"allow": ["*"], "block": []}
128
83
  )
129
84
  expected = resolve_runtime_config(policy_base, task)
130
- assert isinstance(expected, DockerConfig)
85
+ assert isinstance(expected, NetworkPolicyConfig)
131
86
  # Do not inherit extra destinations from a box provisioned for another task.
132
87
  if set(config.allow) != set(expected.allow) or set(config.block) != set(
133
88
  expected.block
@@ -277,8 +232,8 @@ class Agent:
277
232
 
278
233
  Built from an `AgentConfig` alone; `client=`/`interception=` inject live
279
234
  resources to borrow — agents on one endpoint should share one `Client`, and a
280
- live `Interception`'s owner keeps its lifecycle. The harness config's
281
- `runtime` is a *policy*: each `run` provisions a fresh box from it, resolved
235
+ live `Interception`'s owner keeps its lifecycle. The config's `runtime` is a
236
+ *policy*: each `run` provisions a fresh box from it, resolved
282
237
  per task; `run(runtime=...)` places the run into an existing box instead
283
238
  (borrowed boxes are never started or torn down by the run)."""
284
239
 
@@ -296,21 +251,26 @@ class Agent:
296
251
  "AgentConfig.model is unset; an Agent needs a pinned model "
297
252
  "(inside an env the run's own model fills it in)"
298
253
  )
299
- harness_config = config.harness
300
- if harness_config is None:
301
- harness_config = harness_config_type("bash")(id="bash")
254
+ # Resolve the unpinned identity fields into the config: it is the agent's
255
+ # full identity — what stamps onto every trace (`AgentInfo.config`).
256
+ if config.harness is None:
257
+ config = config.model_copy(
258
+ update={"harness": harness_config_type("bash")(id="bash")}
259
+ )
260
+ if config.sampling is None:
261
+ config = config.model_copy(update={"sampling": Sampling()})
302
262
  self.config = config
303
- self.harness = load_harness(harness_config)
263
+ self.harness = load_harness(config.harness)
304
264
  self._owns_client = client is None
305
265
  if self._owns_client:
306
266
  client = resolve_client(config.client or EvalClientConfig())
307
267
  self.ctx = ModelContext(
308
268
  model=config.model,
309
269
  client=client,
310
- sampling=config.sampling if config.sampling is not None else Sampling(),
270
+ sampling=config.sampling,
311
271
  )
312
272
  self._closed = False
313
- self.runtime_config: RuntimeConfig = self.harness.config.runtime
273
+ self.runtime_config: RuntimeConfig = config.runtime
314
274
  self.interception = interception
315
275
  self.limits = RolloutLimits(
316
276
  max_turns=config.max_turns,
@@ -569,6 +529,7 @@ class Agent:
569
529
  else task.data.timeout.harness
570
530
  )
571
531
  return dict(
532
+ agent_config=self.config,
572
533
  harness=self.harness,
573
534
  ctx=self.ctx,
574
535
  runtime_config=runtime_config,
@@ -16,7 +16,7 @@ from verifiers.v1.cli.dashboard.base import live_view
16
16
  from verifiers.v1.cli.output import output_path
17
17
  from verifiers.v1.utils.install import env_name
18
18
  from verifiers.v1.utils.interrupt import cleaning_up
19
- from verifiers.v1.configs.eval import EvalConfig
19
+ from verifiers.v1.configs.cli.eval import EvalConfig
20
20
  from verifiers.v1.env import RunSlot
21
21
  from verifiers.v1.trace import Trace
22
22
  from verifiers.v1.types import Usage
@@ -73,7 +73,7 @@ _MARK = {
73
73
  def _seat_value(config: EvalConfig, read):
74
74
  """A cap as the overview shows it: the declared seats' shared value, the
75
75
  string 'per-seat' when they disagree (caps live on the seats)."""
76
- from verifiers.v1.env import _declared_agent_configs
76
+ from verifiers.v1.configs.env import _declared_agent_configs
77
77
 
78
78
  values = {read(spec) for spec in _declared_agent_configs(config.env).values()}
79
79
  return values.pop() if len(values) == 1 else "per-seat"
@@ -163,8 +163,9 @@ def _warning(config: EvalConfig) -> Text | None:
163
163
  from verifiers.v1.loaders import harness_class
164
164
 
165
165
  if any(
166
- h.runtime.type == "subprocess" and harness_class(h.id).EXECUTES_CODE
167
- for h in config.env.agent_harnesses().values()
166
+ getattr(config.env, role).runtime.type == "subprocess"
167
+ and harness_class(h.id).EXECUTES_CODE
168
+ for role, h in config.env.agent_harnesses().items()
168
169
  ):
169
170
  return Text(
170
171
  "warning Runs on the local system; local files and settings may affect this "
@@ -185,7 +186,7 @@ def overrides(
185
186
  `model_validate(config.toml)`, and that toml is dumped with `exclude_none` (every field), so
186
187
  `model_fields_set` would flag them all. `default` is the reference instance, threaded through
187
188
  recursion so a pinned nested default (`taskset.task.tools`) reads as
188
- unchanged. `skip` holds dotted paths (`harness.runtime.type`)."""
189
+ unchanged. `skip` holds dotted paths (`runtime.type`)."""
189
190
  segments: list[str] = []
190
191
  fields = type(config).model_fields
191
192
  for field in sorted(fields):
@@ -237,9 +238,11 @@ def Overview(config: EvalConfig) -> Table:
237
238
  env_label = f"{env_name(config.env.id)}+{env_label}"
238
239
  # One seat story when every seat resolves the same way (the common case); one
239
240
  # row per seat when they diverge (a judge on its own harness/runtime).
241
+ runtimes = {role: getattr(config.env, role).runtime.type for role in seats}
240
242
  stories = list(
241
243
  dict.fromkeys(
242
- f"{h.name} harness · {h.runtime.type} runtime" for h in seats.values()
244
+ f"{h.name} harness · {runtimes[role]} runtime"
245
+ for role, h in seats.items()
243
246
  )
244
247
  )
245
248
  if len(stories) == 1:
@@ -247,19 +250,23 @@ def Overview(config: EvalConfig) -> Table:
247
250
  else:
248
251
  grid.add_row("env", env_label)
249
252
  for role, h in seats.items():
250
- grid.add_row(f" {role}", f"{h.name} harness · {h.runtime.type} runtime")
253
+ grid.add_row(f" {role}", f"{h.name} harness · {runtimes[role]} runtime")
251
254
  model = f"{config.model} ({sampling})" if sampling else config.model
252
255
  grid.add_row("model", f"{model} via {config.client.base_url}")
253
256
  # Non-default knobs the user set, one row each when non-empty. `escape` the cell: an override
254
257
  # value (or our `[...]`/`{...}` delimiters) can carry Rich markup that would otherwise be
255
- # parsed as styling and dropped. `id` is in the `env` row; harness `runtime.type` too (hidden
256
- # here), but only for the harness — `taskset.task.tools.runtime.type` has no other display.
258
+ # parsed as styling and dropped. `id` is in the `env` row; the seat's `runtime.type` too
259
+ # (hidden here), but only for the seat — `taskset.task.tools.runtime.type` has no other display.
257
260
  if taskset_over := overrides(taskset, skip=frozenset({"id"})):
258
261
  grid.add_row("taskset", escape(" · ".join(taskset_over)))
259
262
  for role, h in seats.items():
260
- if harness_over := overrides(h, skip=frozenset({"id", "runtime.type"})):
263
+ if harness_over := overrides(h, skip=frozenset({"id"})):
261
264
  label = f"{role}.harness" if len(seats) > 1 else "harness"
262
265
  grid.add_row(label, escape(" · ".join(harness_over)))
266
+ runtime = getattr(config.env, role).runtime
267
+ if runtime_over := overrides(runtime, skip=frozenset({"type"})):
268
+ label = f"{role}.runtime" if len(seats) > 1 else "runtime"
269
+ grid.add_row(label, escape(" · ".join(runtime_over)))
263
270
  limits, timeouts = _aligned([_limits(config), _timeouts(config)])
264
271
  grid.add_row("limits", limits)
265
272
  grid.add_row("timeouts", timeouts)
@@ -753,7 +760,8 @@ def _render(
753
760
  now,
754
761
  "/".join(
755
762
  dict.fromkeys(
756
- h.runtime.type for h in config.env.agent_harnesses().values()
763
+ getattr(config.env, role).runtime.type
764
+ for role in config.env.agent_harnesses()
757
765
  )
758
766
  )
759
767
  or "subprocess",
@@ -13,7 +13,7 @@ from rich.table import Table
13
13
  from rich.text import Text
14
14
 
15
15
  from verifiers.v1.cli.dashboard.base import live_view
16
- from verifiers.v1.configs.validate import ValidateConfig
16
+ from verifiers.v1.configs.cli.validate import ValidateConfig
17
17
  from verifiers.v1.utils.format import format_time
18
18
 
19
19
  _STYLE = {
verifiers/v1/cli/debug.py CHANGED
@@ -22,13 +22,13 @@ from verifiers.v1.cli.resolve import (
22
22
  references_config_file,
23
23
  with_positional_taskset,
24
24
  )
25
- from verifiers.v1.configs.debug import DebugConfig
25
+ from verifiers.v1.configs.cli.debug import DebugConfig
26
26
  from verifiers.v1.decorators import invoke
27
27
  from verifiers.v1.utils.compile import resolve_runtime_config
28
28
  from verifiers.v1.runtimes import ProgramResult, Runtime, make_runtime
29
29
  from verifiers.v1.state import state_cls
30
30
  from verifiers.v1.task import Task
31
- from verifiers.v1.trace import Error, Trace, TraceTask
31
+ from verifiers.v1.trace import AgentInfo, Error, Trace, TraceTask
32
32
  from verifiers.v1.utils.interrupt import install_interrupt
33
33
  from verifiers.v1.utils.logging import setup_logging
34
34
 
@@ -202,6 +202,14 @@ async def debug_task(task: Task, config: DebugConfig) -> tuple[Trace, bool]:
202
202
  trace = Trace(
203
203
  task=TraceTask(type=type(task).__name__, data=task.data),
204
204
  state=state_cls(type(task))(),
205
+ # No agent plays here (no model, no harness): the seat records the
206
+ # runtime policy, and the provisioned box lands on it below (id, resolved
207
+ # type, cache status).
208
+ agent=AgentInfo(
209
+ config=vf.AgentConfig(runtime=config.runtime),
210
+ name="debug",
211
+ trainable=False,
212
+ ),
205
213
  )
206
214
  debug = {
207
215
  "task": task_info(task),
@@ -214,7 +222,7 @@ async def debug_task(task: Task, config: DebugConfig) -> tuple[Trace, bool]:
214
222
  resolve_runtime_config(config.runtime, task),
215
223
  name=f"debug-{task.data.idx}-{uuid4().hex[:8]}",
216
224
  )
217
- trace.runtime = runtime.info
225
+ trace.agent.runtime = runtime.info
218
226
  setup_timeout = (
219
227
  config.timeout.setup
220
228
  if config.timeout.setup is not None
@@ -19,7 +19,7 @@ from verifiers.v1.cli.resolve import (
19
19
  )
20
20
  from verifiers.v1.cli.eval.resume import load_resume_config, split_resume
21
21
  from verifiers.v1.cli.eval.runner import run_eval
22
- from verifiers.v1.configs.eval import EvalConfig
22
+ from verifiers.v1.configs.cli.eval import EvalConfig
23
23
 
24
24
  logger = logging.getLogger(__name__)
25
25
 
@@ -22,7 +22,7 @@ from typing import TypeVar
22
22
  from pydantic_core import from_json
23
23
 
24
24
  from verifiers.v1.cli.output import CONFIG_FILE, TRACES_FILE, sniff_episode
25
- from verifiers.v1.configs.eval import EvalConfig
25
+ from verifiers.v1.configs.cli.eval import EvalConfig
26
26
  from verifiers.v1.episode import Episode, WireEpisode
27
27
  from verifiers.v1.trace import WireTrace
28
28
 
@@ -6,7 +6,7 @@ import logging
6
6
  import time
7
7
 
8
8
  from verifiers.v1.clients import ModelContext, resolve_client
9
- from verifiers.v1.configs.eval import EvalConfig
9
+ from verifiers.v1.configs.cli.eval import EvalConfig
10
10
  from verifiers.v1.cli.eval import resume
11
11
  from verifiers.v1.cli.dashboard import dashboard
12
12
  from verifiers.v1.cli.output import (
@@ -108,7 +108,7 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
108
108
  from functools import partial
109
109
 
110
110
  from verifiers.v1.utils.logging import setup_logging
111
- from verifiers.v1.configs.env import pool_serve_kwargs
111
+ from verifiers.v1.configs.cli.env import pool_serve_kwargs
112
112
  from verifiers.v1.serve import EnvClient, env_config_data, serve_env
113
113
 
114
114
  legacy = config.is_legacy
verifiers/v1/cli/init.py CHANGED
@@ -5,7 +5,7 @@ from pathlib import Path
5
5
 
6
6
  from pydantic_config import cli
7
7
 
8
- from verifiers.v1.configs.init import InitConfig
8
+ from verifiers.v1.configs.cli.init import InitConfig
9
9
 
10
10
  USAGE = (
11
11
  "usage: uv run init <name> [--path ./environments] [-T/--add-tool] "
@@ -17,7 +17,7 @@ from pathlib import Path
17
17
  import tomli_w
18
18
  from pydantic import BaseModel, TypeAdapter
19
19
 
20
- from verifiers.v1.configs.eval import EvalConfig
20
+ from verifiers.v1.configs.cli.eval import EvalConfig
21
21
  from verifiers.v1.episode import Episode, WireEpisode
22
22
  from verifiers.v1.trace import Trace
23
23
  from verifiers.v1.utils.aio import run_shielded
@@ -27,9 +27,10 @@ from verifiers.v1.cli.output import (
27
27
  save_config,
28
28
  write_config,
29
29
  )
30
- from verifiers.v1.configs.replay import ReplayConfig
30
+ from verifiers.v1.configs.cli.replay import ReplayConfig
31
31
  from verifiers.v1.state import state_cls
32
32
  from verifiers.v1.task import Task, WireTaskData, task_data_cls
33
+ from verifiers.v1.configs.agent import WireAgentConfig
33
34
  from verifiers.v1.trace import Trace
34
35
  from verifiers.v1.utils.interrupt import install_interrupt
35
36
  from verifiers.v1.utils.logging import setup_logging
@@ -86,7 +87,9 @@ async def run_replay(config: ReplayConfig, source: Path, out: Path) -> list[Trac
86
87
  # An episode may hold no traces (its env hooks failed before any agent ran);
87
88
  # there's nothing to re-score, so it drops out in the flatten. Each kept trace
88
89
  # remembers its episode's env identity — the replay output re-stamps it.
89
- episodes = read_episodes(source, Trace[WireTaskData, state_cls(task_cls)])
90
+ episodes = read_episodes(
91
+ source, Trace[WireTaskData, state_cls(task_cls), WireAgentConfig]
92
+ )
90
93
  sourced = [(trace, e.env) for e in episodes for trace in e.traces]
91
94
  if config.num_traces is not None:
92
95
  sourced = sourced[: config.num_traces]
verifiers/v1/cli/serve.py CHANGED
@@ -13,8 +13,8 @@ from verifiers.v1.cli.resolve import (
13
13
  references_config_file,
14
14
  with_positional_taskset,
15
15
  )
16
- from verifiers.v1.configs.serve import ServeConfig
17
- from verifiers.v1.configs.env import pool_serve_kwargs
16
+ from verifiers.v1.configs.cli.serve import ServeConfig
17
+ from verifiers.v1.configs.cli.env import pool_serve_kwargs
18
18
  from verifiers.v1.serve import serve_env
19
19
 
20
20
  USAGE = "usage: uv run serve [<taskset-id>] [--env.id <id>] [--id <env-id> (legacy)] [options] [@ file.toml]"
@@ -18,7 +18,7 @@ from verifiers.v1.cli.resolve import (
18
18
  references_config_file,
19
19
  with_positional_taskset,
20
20
  )
21
- from verifiers.v1.configs.validate import ValidateConfig
21
+ from verifiers.v1.configs.cli.validate import ValidateConfig
22
22
  from verifiers.v1.decorators import invoke
23
23
  from verifiers.v1.utils.compile import resolve_runtime_config
24
24
  from verifiers.v1.runtimes import make_runtime
@@ -1,7 +1,4 @@
1
- from verifiers.v1.configs.eval import EvalConfig
2
- from verifiers.v1.configs.debug import DebugConfig
3
- from verifiers.v1.configs.init import InitConfig
4
- from verifiers.v1.configs.serve import ServeConfig
5
- from verifiers.v1.configs.validate import ValidateConfig
6
-
7
- __all__ = ["DebugConfig", "EvalConfig", "InitConfig", "ServeConfig", "ValidateConfig"]
1
+ """The config tree, separated from the logic it configures: one module per
2
+ domain (`configs.agent`, `configs.harness`, ...) plus the CLI entrypoints' run
3
+ configs under `configs.cli`. Config modules import only types and other configs,
4
+ so any module — the trace record included — can embed them without cycles."""
@@ -0,0 +1,77 @@
1
+ """One env agent's config: who plays the seat, and its per-run caps."""
2
+
3
+ from pydantic import SerializeAsAny, model_validator
4
+ from pydantic_config import BaseConfig
5
+
6
+ from verifiers.v1.clients import ClientConfig
7
+ from verifiers.v1.configs.harness import HarnessConfig, WireHarnessConfig
8
+ from verifiers.v1.configs.retries import RetryConfig
9
+ from verifiers.v1.runtimes import RuntimeConfig, SubprocessConfig
10
+ from verifiers.v1.types import SamplingConfig
11
+
12
+
13
+ class TimeoutConfig(BaseConfig):
14
+ """Per-agent wall-clock timeouts per rollout stage, in seconds (None = no
15
+ limit); each stage falls back to the task's own `TaskTimeout` when unset. An
16
+ interaction's rollout budget is cumulative across its active harness segments
17
+ and pauses while the caller computes the next user turn."""
18
+
19
+ setup: float | None = None # one shared budget: task setup + provisioning
20
+ rollout: float | None = None
21
+ finalize: float | None = None
22
+ scoring: float | None = None
23
+
24
+
25
+ class AgentConfig(BaseConfig):
26
+ """One env agent: who plays it, and its per-run caps. It pins only what
27
+ makes it a different actor; everything unpinned falls back — the model context
28
+ to the run's own, the harness to the taskset's default."""
29
+
30
+ harness: SerializeAsAny[HarnessConfig] | None = None
31
+ """The agent's program (None = the taskset's default harness)."""
32
+ runtime: RuntimeConfig = SubprocessConfig()
33
+ """Runtime for the harness program — the policy each run provisions its box
34
+ from; tool servers choose their placement separately."""
35
+ model: str | None = None
36
+ """Model id (None = the run's model, i.e. the policy under evaluation/training)."""
37
+ client: ClientConfig | None = None
38
+ """Endpoint override (None = the run's client)."""
39
+ sampling: SamplingConfig | None = None
40
+ """Sampling override (None = the run's sampling)."""
41
+ timeout: TimeoutConfig = TimeoutConfig()
42
+ retries: RetryConfig = RetryConfig()
43
+ """Whole-run retries: rerun this agent's rollout while its trace ends with a
44
+ retryable error (never into a borrowed box)."""
45
+ max_turns: int | None = None
46
+ """Max model turns per run (None = no limit). Framework-enforced (the
47
+ interception server refuses turns past it), so it applies to any harness."""
48
+ max_input_tokens: int | None = None
49
+ max_output_tokens: int | None = None
50
+ max_total_tokens: int | None = None
51
+ """Token caps per run (None = no limit); framework-enforced between turns."""
52
+
53
+ @model_validator(mode="before")
54
+ @classmethod
55
+ def _resolve_harness(cls, data):
56
+ """Narrow a pinned `harness` to its concrete config type by `id` (absent
57
+ stays None = the taskset's default). The lazy import keeps class-body
58
+ `AgentConfig()` defaults constructible while this module initializes."""
59
+ if isinstance(data, dict) and data.get("harness") is not None:
60
+ from verifiers.v1.loaders import harness_config_type, narrow_plugin_field
61
+
62
+ narrow_plugin_field(data, "harness", harness_config_type, "bash")
63
+ return data
64
+
65
+
66
+ class WireAgentConfig(AgentConfig):
67
+ """Wire form for trace records: parses without resolving the harness plugin,
68
+ so records round-trip anywhere — the knobs stay readable on the extra-allow
69
+ `WireHarnessConfig` (see `WireTaskData`)."""
70
+
71
+ harness: SerializeAsAny[WireHarnessConfig] | None = None
72
+
73
+ @model_validator(mode="before")
74
+ @classmethod
75
+ def _resolve_harness(cls, data):
76
+ """Override: a record read resolves no plugins."""
77
+ return data
@@ -0,0 +1,9 @@
1
+ """The CLI entrypoints' run configs — each `uv run <cmd>`'s parsed tree."""
2
+
3
+ from verifiers.v1.configs.cli.eval import EvalConfig
4
+ from verifiers.v1.configs.cli.debug import DebugConfig
5
+ from verifiers.v1.configs.cli.init import InitConfig
6
+ from verifiers.v1.configs.cli.serve import ServeConfig
7
+ from verifiers.v1.configs.cli.validate import ValidateConfig
8
+
9
+ __all__ = ["DebugConfig", "EvalConfig", "InitConfig", "ServeConfig", "ValidateConfig"]
@@ -6,9 +6,9 @@ from uuid import uuid4
6
6
  from pydantic import AliasChoices, Field, SerializeAsAny, model_validator
7
7
  from pydantic_config import BaseConfig
8
8
 
9
- from verifiers.v1.configs.validate import CheckTimeoutConfig
9
+ from verifiers.v1.configs.cli.validate import CheckTimeoutConfig
10
10
  from verifiers.v1.runtimes import DockerConfig, RuntimeConfig
11
- from verifiers.v1.taskset import TasksetConfig
11
+ from verifiers.v1.configs.taskset import TasksetConfig
12
12
 
13
13
 
14
14
  class DebugConfig(BaseConfig):