verifiers 0.2.2.dev23__py3-none-any.whl → 0.2.2.dev25__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/__init__.py +14 -33
- verifiers/v1/agent.py +34 -73
- verifiers/v1/cli/dashboard/eval.py +19 -11
- verifiers/v1/cli/dashboard/validate.py +1 -1
- verifiers/v1/cli/debug.py +11 -3
- verifiers/v1/cli/eval/main.py +1 -1
- verifiers/v1/cli/eval/resume.py +1 -1
- verifiers/v1/cli/eval/runner.py +2 -2
- verifiers/v1/cli/init.py +1 -1
- verifiers/v1/cli/output.py +1 -1
- verifiers/v1/cli/replay.py +5 -2
- verifiers/v1/cli/serve.py +2 -2
- verifiers/v1/cli/validate.py +1 -1
- verifiers/v1/configs/__init__.py +4 -7
- verifiers/v1/configs/agent.py +77 -0
- verifiers/v1/configs/cli/__init__.py +9 -0
- verifiers/v1/configs/{debug.py → cli/debug.py} +2 -2
- verifiers/v1/configs/cli/env.py +167 -0
- verifiers/v1/configs/{eval.py → cli/eval.py} +1 -1
- verifiers/v1/configs/{replay.py → cli/replay.py} +1 -1
- verifiers/v1/configs/{serve.py → cli/serve.py} +1 -1
- verifiers/v1/configs/{validate.py → cli/validate.py} +1 -1
- verifiers/v1/configs/env.py +139 -150
- verifiers/v1/configs/harness.py +42 -0
- verifiers/v1/configs/judge.py +73 -0
- verifiers/v1/configs/retries.py +17 -0
- verifiers/v1/configs/task.py +32 -0
- verifiers/v1/configs/taskset.py +20 -0
- verifiers/v1/env.py +12 -155
- verifiers/v1/envs/agentic_judge/env.py +5 -5
- verifiers/v1/envs/single_agent/env.py +4 -2
- verifiers/v1/episode.py +7 -5
- verifiers/v1/gepa/config.py +2 -2
- verifiers/v1/gepa/reflection.py +2 -1
- verifiers/v1/harness.py +5 -39
- verifiers/v1/harnesses/bash/harness.py +2 -1
- verifiers/v1/harnesses/claude_code/harness.py +2 -1
- verifiers/v1/harnesses/codex/harness.py +2 -1
- verifiers/v1/harnesses/kimi_code/harness.py +2 -1
- verifiers/v1/harnesses/mini_swe_agent/harness.py +2 -1
- verifiers/v1/harnesses/null/harness.py +2 -1
- verifiers/v1/harnesses/pi/harness.py +2 -1
- verifiers/v1/harnesses/pool/harness.py +2 -1
- verifiers/v1/harnesses/rlm/harness.py +2 -1
- verifiers/v1/harnesses/terminus_2/harness.py +2 -1
- verifiers/v1/judge.py +7 -66
- verifiers/v1/judges/reference.py +1 -1
- verifiers/v1/judges/rubric.py +2 -7
- verifiers/v1/loaders.py +8 -4
- verifiers/v1/mcp/launch.py +7 -7
- verifiers/v1/push.py +1 -1
- verifiers/v1/retries.py +1 -15
- verifiers/v1/rollout.py +7 -7
- verifiers/v1/runtimes/__init__.py +2 -0
- verifiers/v1/runtimes/base.py +43 -3
- verifiers/v1/runtimes/docker/__init__.py +8 -35
- verifiers/v1/runtimes/prime.py +57 -4
- verifiers/v1/serve/client.py +3 -4
- verifiers/v1/serve/pool.py +1 -1
- verifiers/v1/serve/server.py +1 -1
- verifiers/v1/serve/types.py +4 -6
- verifiers/v1/task.py +5 -29
- verifiers/v1/taskset.py +2 -16
- verifiers/v1/tasksets/harbor/taskset.py +2 -1
- verifiers/v1/tasksets/lean/taskset.py +4 -2
- verifiers/v1/trace.py +34 -21
- verifiers/v1/utils/compile.py +8 -14
- {verifiers-0.2.2.dev23.dist-info → verifiers-0.2.2.dev25.dist-info}/METADATA +2 -2
- {verifiers-0.2.2.dev23.dist-info → verifiers-0.2.2.dev25.dist-info}/RECORD +73 -65
- /verifiers/v1/configs/{init.py → cli/init.py} +0 -0
- {verifiers-0.2.2.dev23.dist-info → verifiers-0.2.2.dev25.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev23.dist-info → verifiers-0.2.2.dev25.dist-info}/entry_points.txt +0 -0
- {verifiers-0.2.2.dev23.dist-info → verifiers-0.2.2.dev25.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/__init__.py
CHANGED
|
@@ -13,25 +13,16 @@ from verifiers.v1.clients import (
|
|
|
13
13
|
resolve_client,
|
|
14
14
|
)
|
|
15
15
|
from verifiers.v1.decorators import metric, reward, stop, tool
|
|
16
|
-
from verifiers.v1.agent import
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
Agents,
|
|
20
|
-
Interaction,
|
|
21
|
-
Segment,
|
|
22
|
-
make_agent,
|
|
23
|
-
)
|
|
24
|
-
from verifiers.v1.configs.env import (
|
|
16
|
+
from verifiers.v1.configs.agent import AgentConfig
|
|
17
|
+
from verifiers.v1.agent import Agent, Agents, Interaction, Segment, make_agent
|
|
18
|
+
from verifiers.v1.configs.cli.env import (
|
|
25
19
|
ElasticPoolConfig,
|
|
26
20
|
EnvServerConfig,
|
|
27
21
|
StaticPoolConfig,
|
|
28
22
|
pool_serve_kwargs,
|
|
29
23
|
)
|
|
30
|
-
from verifiers.v1.env import
|
|
31
|
-
|
|
32
|
-
Env,
|
|
33
|
-
default_agent_harness,
|
|
34
|
-
)
|
|
24
|
+
from verifiers.v1.configs.env import EnvConfig, default_agent_harness
|
|
25
|
+
from verifiers.v1.env import Env
|
|
35
26
|
from verifiers.v1.envs.single_agent import SingleAgentEnv, SingleAgentEnvConfig
|
|
36
27
|
from verifiers.v1.errors import (
|
|
37
28
|
EnvError,
|
|
@@ -44,15 +35,10 @@ from verifiers.v1.errors import (
|
|
|
44
35
|
ToolsetError,
|
|
45
36
|
TunnelError,
|
|
46
37
|
)
|
|
47
|
-
from verifiers.v1.harness import
|
|
48
|
-
from verifiers.v1.
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
JudgeResponse,
|
|
52
|
-
Judges,
|
|
53
|
-
JudgeSamplingConfig,
|
|
54
|
-
JudgeView,
|
|
55
|
-
)
|
|
38
|
+
from verifiers.v1.configs.harness import HarnessConfig
|
|
39
|
+
from verifiers.v1.harness import Harness
|
|
40
|
+
from verifiers.v1.configs.judge import JudgeConfig, JudgeSamplingConfig, Judges
|
|
41
|
+
from verifiers.v1.judge import Judge, JudgeResponse, JudgeView
|
|
56
42
|
from verifiers.v1.judges import (
|
|
57
43
|
ReferenceJudge,
|
|
58
44
|
ReferenceJudgeConfig,
|
|
@@ -86,7 +72,7 @@ from verifiers.v1.scoring import (
|
|
|
86
72
|
read_answer_file_or_last_reply as read_answer_file_or_last_reply,
|
|
87
73
|
verify_boxed_math_answer as verify_boxed_math_answer,
|
|
88
74
|
)
|
|
89
|
-
from verifiers.v1.retries import RetryConfig
|
|
75
|
+
from verifiers.v1.configs.retries import RetryConfig
|
|
90
76
|
from verifiers.v1.utils.git import (
|
|
91
77
|
PATCH_CAP_BYTES as PATCH_CAP_BYTES,
|
|
92
78
|
capture_patch as capture_patch,
|
|
@@ -102,15 +88,10 @@ from verifiers.v1.runtimes import (
|
|
|
102
88
|
SubprocessConfig,
|
|
103
89
|
)
|
|
104
90
|
from verifiers.v1.state import State, StateT
|
|
105
|
-
from verifiers.v1.task import
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
TaskResources,
|
|
110
|
-
TaskTimeout,
|
|
111
|
-
WireTaskData,
|
|
112
|
-
)
|
|
113
|
-
from verifiers.v1.taskset import Taskset, TasksetConfig
|
|
91
|
+
from verifiers.v1.configs.task import TaskConfig
|
|
92
|
+
from verifiers.v1.task import Task, TaskData, TaskResources, TaskTimeout, WireTaskData
|
|
93
|
+
from verifiers.v1.configs.taskset import TasksetConfig
|
|
94
|
+
from verifiers.v1.taskset import Taskset
|
|
114
95
|
from verifiers.v1.mcp import (
|
|
115
96
|
Toolset,
|
|
116
97
|
SharedToolsetConfig,
|
verifiers/v1/agent.py
CHANGED
|
@@ -14,23 +14,21 @@ from contextlib import asynccontextmanager, nullcontext
|
|
|
14
14
|
from dataclasses import dataclass
|
|
15
15
|
from typing import AsyncIterator
|
|
16
16
|
|
|
17
|
-
from pydantic import SerializeAsAny, model_validator
|
|
18
|
-
from pydantic_config import BaseConfig
|
|
19
17
|
|
|
18
|
+
from verifiers.v1.configs.agent import AgentConfig, TimeoutConfig
|
|
20
19
|
from verifiers.v1.clients import (
|
|
21
20
|
Client,
|
|
22
|
-
ClientConfig,
|
|
23
21
|
EvalClientConfig,
|
|
24
22
|
ModelContext,
|
|
25
23
|
resolve_client,
|
|
26
24
|
)
|
|
27
|
-
from verifiers.v1.harness import Harness
|
|
25
|
+
from verifiers.v1.harness import Harness
|
|
28
26
|
from verifiers.v1.interception import Interception, InterceptionServer
|
|
29
27
|
from verifiers.v1.mcp import SharedToolServer
|
|
30
|
-
from verifiers.v1.retries import
|
|
28
|
+
from verifiers.v1.retries import backoff, trace_should_retry
|
|
31
29
|
from verifiers.v1.rollout import RolloutRun, _as_messages
|
|
32
30
|
from verifiers.v1.runtimes import (
|
|
33
|
-
|
|
31
|
+
NetworkPolicyConfig,
|
|
34
32
|
Runtime,
|
|
35
33
|
RuntimeConfig,
|
|
36
34
|
SubprocessConfig,
|
|
@@ -44,7 +42,6 @@ from verifiers.v1.types import (
|
|
|
44
42
|
AssistantMessage,
|
|
45
43
|
Messages,
|
|
46
44
|
Sampling,
|
|
47
|
-
SamplingConfig,
|
|
48
45
|
ToolMessage,
|
|
49
46
|
UserMessage,
|
|
50
47
|
)
|
|
@@ -54,57 +51,9 @@ from verifiers.v1.utils.compile import (
|
|
|
54
51
|
validate_pairing,
|
|
55
52
|
)
|
|
56
53
|
|
|
57
|
-
|
|
58
|
-
|
|
54
|
+
__all__ = ["Agent", "AgentConfig", "Agents", "TimeoutConfig", "make_agent"]
|
|
59
55
|
|
|
60
|
-
|
|
61
|
-
"""Per-agent wall-clock timeouts per rollout stage, in seconds (None = no
|
|
62
|
-
limit); each stage falls back to the task's own `TaskTimeout` when unset. An
|
|
63
|
-
interaction's rollout budget is cumulative across its active harness segments
|
|
64
|
-
and pauses while the caller computes the next user turn."""
|
|
65
|
-
|
|
66
|
-
setup: float | None = None # one shared budget: task setup + provisioning
|
|
67
|
-
rollout: float | None = None
|
|
68
|
-
finalize: float | None = None
|
|
69
|
-
scoring: float | None = None
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
class AgentConfig(BaseConfig):
|
|
73
|
-
"""One env agent: who plays it, and its per-run caps. It pins only what
|
|
74
|
-
makes it a different actor; everything unpinned falls back — the model context
|
|
75
|
-
to the run's own, the harness to the taskset's default."""
|
|
76
|
-
|
|
77
|
-
harness: SerializeAsAny[HarnessConfig] | None = None
|
|
78
|
-
"""The agent's program + runtime policy (None = the taskset's default harness)."""
|
|
79
|
-
model: str | None = None
|
|
80
|
-
"""Model id (None = the run's model, i.e. the policy under evaluation/training)."""
|
|
81
|
-
client: ClientConfig | None = None
|
|
82
|
-
"""Endpoint override (None = the run's client)."""
|
|
83
|
-
sampling: SamplingConfig | None = None
|
|
84
|
-
"""Sampling override (None = the run's sampling)."""
|
|
85
|
-
timeout: TimeoutConfig = TimeoutConfig()
|
|
86
|
-
retries: RetryConfig = RetryConfig()
|
|
87
|
-
"""Whole-run retries: rerun this agent's rollout while its trace ends with a
|
|
88
|
-
retryable error (never into a borrowed box)."""
|
|
89
|
-
max_turns: int | None = None
|
|
90
|
-
"""Max model turns per run (None = no limit). Framework-enforced (the
|
|
91
|
-
interception server refuses turns past it), so it applies to any harness."""
|
|
92
|
-
max_input_tokens: int | None = None
|
|
93
|
-
max_output_tokens: int | None = None
|
|
94
|
-
max_total_tokens: int | None = None
|
|
95
|
-
"""Token caps per run (None = no limit); framework-enforced between turns."""
|
|
96
|
-
|
|
97
|
-
@model_validator(mode="before")
|
|
98
|
-
@classmethod
|
|
99
|
-
def _resolve_harness(cls, data):
|
|
100
|
-
"""Narrow a pinned `harness` to its concrete config type by `id` (absent
|
|
101
|
-
stays None = the taskset's default). The lazy import keeps class-body
|
|
102
|
-
`AgentConfig()` defaults constructible while this module initializes."""
|
|
103
|
-
if isinstance(data, dict) and data.get("harness") is not None:
|
|
104
|
-
from verifiers.v1.loaders import harness_config_type, narrow_plugin_field
|
|
105
|
-
|
|
106
|
-
narrow_plugin_field(data, "harness", harness_config_type, "bash")
|
|
107
|
-
return data
|
|
56
|
+
logger = logging.getLogger(__name__)
|
|
108
57
|
|
|
109
58
|
|
|
110
59
|
def _check_borrowed_placement(
|
|
@@ -114,20 +63,26 @@ def _check_borrowed_placement(
|
|
|
114
63
|
be honored. Reject requirements that cannot be applied to the running box; an
|
|
115
64
|
image mismatch on a container only warns, since sharing its world is the point."""
|
|
116
65
|
task_policy = "*" not in task.data.network_allow or bool(task.data.network_block)
|
|
117
|
-
|
|
118
|
-
if task_policy or (
|
|
66
|
+
base_policy = base_config if isinstance(base_config, NetworkPolicyConfig) else None
|
|
67
|
+
if task_policy or (base_policy is not None and base_policy.network_restricted):
|
|
119
68
|
config = runtime.config
|
|
120
|
-
if not isinstance(config,
|
|
69
|
+
if not isinstance(config, NetworkPolicyConfig):
|
|
121
70
|
raise ValueError(
|
|
122
|
-
f"task {task.data.idx!r} requires a
|
|
123
|
-
f"borrowed runtime {runtime.name!r}
|
|
71
|
+
f"task {task.data.idx!r} requires a framework-aware network policy, "
|
|
72
|
+
f"but borrowed runtime {runtime.name!r} does not support one; use "
|
|
124
73
|
"agent.provision(task)"
|
|
125
74
|
)
|
|
126
|
-
|
|
127
|
-
|
|
75
|
+
if base_policy is not None and type(config) is not type(base_policy):
|
|
76
|
+
raise ValueError(
|
|
77
|
+
f"the configured {base_policy.type} network policy cannot be applied "
|
|
78
|
+
f"to borrowed {config.type} runtime {runtime.name!r}; use "
|
|
79
|
+
"agent.provision(task)"
|
|
80
|
+
)
|
|
81
|
+
policy_base = base_policy or config.model_copy(
|
|
82
|
+
update={"allow": ["*"], "block": []}
|
|
128
83
|
)
|
|
129
84
|
expected = resolve_runtime_config(policy_base, task)
|
|
130
|
-
assert isinstance(expected,
|
|
85
|
+
assert isinstance(expected, NetworkPolicyConfig)
|
|
131
86
|
# Do not inherit extra destinations from a box provisioned for another task.
|
|
132
87
|
if set(config.allow) != set(expected.allow) or set(config.block) != set(
|
|
133
88
|
expected.block
|
|
@@ -277,8 +232,8 @@ class Agent:
|
|
|
277
232
|
|
|
278
233
|
Built from an `AgentConfig` alone; `client=`/`interception=` inject live
|
|
279
234
|
resources to borrow — agents on one endpoint should share one `Client`, and a
|
|
280
|
-
live `Interception`'s owner keeps its lifecycle. The
|
|
281
|
-
|
|
235
|
+
live `Interception`'s owner keeps its lifecycle. The config's `runtime` is a
|
|
236
|
+
*policy*: each `run` provisions a fresh box from it, resolved
|
|
282
237
|
per task; `run(runtime=...)` places the run into an existing box instead
|
|
283
238
|
(borrowed boxes are never started or torn down by the run)."""
|
|
284
239
|
|
|
@@ -296,21 +251,26 @@ class Agent:
|
|
|
296
251
|
"AgentConfig.model is unset; an Agent needs a pinned model "
|
|
297
252
|
"(inside an env the run's own model fills it in)"
|
|
298
253
|
)
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
254
|
+
# Resolve the unpinned identity fields into the config: it is the agent's
|
|
255
|
+
# full identity — what stamps onto every trace (`AgentInfo.config`).
|
|
256
|
+
if config.harness is None:
|
|
257
|
+
config = config.model_copy(
|
|
258
|
+
update={"harness": harness_config_type("bash")(id="bash")}
|
|
259
|
+
)
|
|
260
|
+
if config.sampling is None:
|
|
261
|
+
config = config.model_copy(update={"sampling": Sampling()})
|
|
302
262
|
self.config = config
|
|
303
|
-
self.harness = load_harness(
|
|
263
|
+
self.harness = load_harness(config.harness)
|
|
304
264
|
self._owns_client = client is None
|
|
305
265
|
if self._owns_client:
|
|
306
266
|
client = resolve_client(config.client or EvalClientConfig())
|
|
307
267
|
self.ctx = ModelContext(
|
|
308
268
|
model=config.model,
|
|
309
269
|
client=client,
|
|
310
|
-
sampling=config.sampling
|
|
270
|
+
sampling=config.sampling,
|
|
311
271
|
)
|
|
312
272
|
self._closed = False
|
|
313
|
-
self.runtime_config: RuntimeConfig =
|
|
273
|
+
self.runtime_config: RuntimeConfig = config.runtime
|
|
314
274
|
self.interception = interception
|
|
315
275
|
self.limits = RolloutLimits(
|
|
316
276
|
max_turns=config.max_turns,
|
|
@@ -569,6 +529,7 @@ class Agent:
|
|
|
569
529
|
else task.data.timeout.harness
|
|
570
530
|
)
|
|
571
531
|
return dict(
|
|
532
|
+
agent_config=self.config,
|
|
572
533
|
harness=self.harness,
|
|
573
534
|
ctx=self.ctx,
|
|
574
535
|
runtime_config=runtime_config,
|
|
@@ -16,7 +16,7 @@ from verifiers.v1.cli.dashboard.base import live_view
|
|
|
16
16
|
from verifiers.v1.cli.output import output_path
|
|
17
17
|
from verifiers.v1.utils.install import env_name
|
|
18
18
|
from verifiers.v1.utils.interrupt import cleaning_up
|
|
19
|
-
from verifiers.v1.configs.eval import EvalConfig
|
|
19
|
+
from verifiers.v1.configs.cli.eval import EvalConfig
|
|
20
20
|
from verifiers.v1.env import RunSlot
|
|
21
21
|
from verifiers.v1.trace import Trace
|
|
22
22
|
from verifiers.v1.types import Usage
|
|
@@ -73,7 +73,7 @@ _MARK = {
|
|
|
73
73
|
def _seat_value(config: EvalConfig, read):
|
|
74
74
|
"""A cap as the overview shows it: the declared seats' shared value, the
|
|
75
75
|
string 'per-seat' when they disagree (caps live on the seats)."""
|
|
76
|
-
from verifiers.v1.env import _declared_agent_configs
|
|
76
|
+
from verifiers.v1.configs.env import _declared_agent_configs
|
|
77
77
|
|
|
78
78
|
values = {read(spec) for spec in _declared_agent_configs(config.env).values()}
|
|
79
79
|
return values.pop() if len(values) == 1 else "per-seat"
|
|
@@ -163,8 +163,9 @@ def _warning(config: EvalConfig) -> Text | None:
|
|
|
163
163
|
from verifiers.v1.loaders import harness_class
|
|
164
164
|
|
|
165
165
|
if any(
|
|
166
|
-
|
|
167
|
-
|
|
166
|
+
getattr(config.env, role).runtime.type == "subprocess"
|
|
167
|
+
and harness_class(h.id).EXECUTES_CODE
|
|
168
|
+
for role, h in config.env.agent_harnesses().items()
|
|
168
169
|
):
|
|
169
170
|
return Text(
|
|
170
171
|
"warning Runs on the local system; local files and settings may affect this "
|
|
@@ -185,7 +186,7 @@ def overrides(
|
|
|
185
186
|
`model_validate(config.toml)`, and that toml is dumped with `exclude_none` (every field), so
|
|
186
187
|
`model_fields_set` would flag them all. `default` is the reference instance, threaded through
|
|
187
188
|
recursion so a pinned nested default (`taskset.task.tools`) reads as
|
|
188
|
-
unchanged. `skip` holds dotted paths (`
|
|
189
|
+
unchanged. `skip` holds dotted paths (`runtime.type`)."""
|
|
189
190
|
segments: list[str] = []
|
|
190
191
|
fields = type(config).model_fields
|
|
191
192
|
for field in sorted(fields):
|
|
@@ -237,9 +238,11 @@ def Overview(config: EvalConfig) -> Table:
|
|
|
237
238
|
env_label = f"{env_name(config.env.id)}+{env_label}"
|
|
238
239
|
# One seat story when every seat resolves the same way (the common case); one
|
|
239
240
|
# row per seat when they diverge (a judge on its own harness/runtime).
|
|
241
|
+
runtimes = {role: getattr(config.env, role).runtime.type for role in seats}
|
|
240
242
|
stories = list(
|
|
241
243
|
dict.fromkeys(
|
|
242
|
-
f"{h.name} harness · {
|
|
244
|
+
f"{h.name} harness · {runtimes[role]} runtime"
|
|
245
|
+
for role, h in seats.items()
|
|
243
246
|
)
|
|
244
247
|
)
|
|
245
248
|
if len(stories) == 1:
|
|
@@ -247,19 +250,23 @@ def Overview(config: EvalConfig) -> Table:
|
|
|
247
250
|
else:
|
|
248
251
|
grid.add_row("env", env_label)
|
|
249
252
|
for role, h in seats.items():
|
|
250
|
-
grid.add_row(f" {role}", f"{h.name} harness · {
|
|
253
|
+
grid.add_row(f" {role}", f"{h.name} harness · {runtimes[role]} runtime")
|
|
251
254
|
model = f"{config.model} ({sampling})" if sampling else config.model
|
|
252
255
|
grid.add_row("model", f"{model} via {config.client.base_url}")
|
|
253
256
|
# Non-default knobs the user set, one row each when non-empty. `escape` the cell: an override
|
|
254
257
|
# value (or our `[...]`/`{...}` delimiters) can carry Rich markup that would otherwise be
|
|
255
|
-
# parsed as styling and dropped. `id` is in the `env` row;
|
|
256
|
-
# here), but only for the
|
|
258
|
+
# parsed as styling and dropped. `id` is in the `env` row; the seat's `runtime.type` too
|
|
259
|
+
# (hidden here), but only for the seat — `taskset.task.tools.runtime.type` has no other display.
|
|
257
260
|
if taskset_over := overrides(taskset, skip=frozenset({"id"})):
|
|
258
261
|
grid.add_row("taskset", escape(" · ".join(taskset_over)))
|
|
259
262
|
for role, h in seats.items():
|
|
260
|
-
if harness_over := overrides(h, skip=frozenset({"id"
|
|
263
|
+
if harness_over := overrides(h, skip=frozenset({"id"})):
|
|
261
264
|
label = f"{role}.harness" if len(seats) > 1 else "harness"
|
|
262
265
|
grid.add_row(label, escape(" · ".join(harness_over)))
|
|
266
|
+
runtime = getattr(config.env, role).runtime
|
|
267
|
+
if runtime_over := overrides(runtime, skip=frozenset({"type"})):
|
|
268
|
+
label = f"{role}.runtime" if len(seats) > 1 else "runtime"
|
|
269
|
+
grid.add_row(label, escape(" · ".join(runtime_over)))
|
|
263
270
|
limits, timeouts = _aligned([_limits(config), _timeouts(config)])
|
|
264
271
|
grid.add_row("limits", limits)
|
|
265
272
|
grid.add_row("timeouts", timeouts)
|
|
@@ -753,7 +760,8 @@ def _render(
|
|
|
753
760
|
now,
|
|
754
761
|
"/".join(
|
|
755
762
|
dict.fromkeys(
|
|
756
|
-
|
|
763
|
+
getattr(config.env, role).runtime.type
|
|
764
|
+
for role in config.env.agent_harnesses()
|
|
757
765
|
)
|
|
758
766
|
)
|
|
759
767
|
or "subprocess",
|
|
@@ -13,7 +13,7 @@ from rich.table import Table
|
|
|
13
13
|
from rich.text import Text
|
|
14
14
|
|
|
15
15
|
from verifiers.v1.cli.dashboard.base import live_view
|
|
16
|
-
from verifiers.v1.configs.validate import ValidateConfig
|
|
16
|
+
from verifiers.v1.configs.cli.validate import ValidateConfig
|
|
17
17
|
from verifiers.v1.utils.format import format_time
|
|
18
18
|
|
|
19
19
|
_STYLE = {
|
verifiers/v1/cli/debug.py
CHANGED
|
@@ -22,13 +22,13 @@ from verifiers.v1.cli.resolve import (
|
|
|
22
22
|
references_config_file,
|
|
23
23
|
with_positional_taskset,
|
|
24
24
|
)
|
|
25
|
-
from verifiers.v1.configs.debug import DebugConfig
|
|
25
|
+
from verifiers.v1.configs.cli.debug import DebugConfig
|
|
26
26
|
from verifiers.v1.decorators import invoke
|
|
27
27
|
from verifiers.v1.utils.compile import resolve_runtime_config
|
|
28
28
|
from verifiers.v1.runtimes import ProgramResult, Runtime, make_runtime
|
|
29
29
|
from verifiers.v1.state import state_cls
|
|
30
30
|
from verifiers.v1.task import Task
|
|
31
|
-
from verifiers.v1.trace import Error, Trace, TraceTask
|
|
31
|
+
from verifiers.v1.trace import AgentInfo, Error, Trace, TraceTask
|
|
32
32
|
from verifiers.v1.utils.interrupt import install_interrupt
|
|
33
33
|
from verifiers.v1.utils.logging import setup_logging
|
|
34
34
|
|
|
@@ -202,6 +202,14 @@ async def debug_task(task: Task, config: DebugConfig) -> tuple[Trace, bool]:
|
|
|
202
202
|
trace = Trace(
|
|
203
203
|
task=TraceTask(type=type(task).__name__, data=task.data),
|
|
204
204
|
state=state_cls(type(task))(),
|
|
205
|
+
# No agent plays here (no model, no harness): the seat records the
|
|
206
|
+
# runtime policy, and the provisioned box lands on it below (id, resolved
|
|
207
|
+
# type, cache status).
|
|
208
|
+
agent=AgentInfo(
|
|
209
|
+
config=vf.AgentConfig(runtime=config.runtime),
|
|
210
|
+
name="debug",
|
|
211
|
+
trainable=False,
|
|
212
|
+
),
|
|
205
213
|
)
|
|
206
214
|
debug = {
|
|
207
215
|
"task": task_info(task),
|
|
@@ -214,7 +222,7 @@ async def debug_task(task: Task, config: DebugConfig) -> tuple[Trace, bool]:
|
|
|
214
222
|
resolve_runtime_config(config.runtime, task),
|
|
215
223
|
name=f"debug-{task.data.idx}-{uuid4().hex[:8]}",
|
|
216
224
|
)
|
|
217
|
-
trace.runtime = runtime.info
|
|
225
|
+
trace.agent.runtime = runtime.info
|
|
218
226
|
setup_timeout = (
|
|
219
227
|
config.timeout.setup
|
|
220
228
|
if config.timeout.setup is not None
|
verifiers/v1/cli/eval/main.py
CHANGED
|
@@ -19,7 +19,7 @@ from verifiers.v1.cli.resolve import (
|
|
|
19
19
|
)
|
|
20
20
|
from verifiers.v1.cli.eval.resume import load_resume_config, split_resume
|
|
21
21
|
from verifiers.v1.cli.eval.runner import run_eval
|
|
22
|
-
from verifiers.v1.configs.eval import EvalConfig
|
|
22
|
+
from verifiers.v1.configs.cli.eval import EvalConfig
|
|
23
23
|
|
|
24
24
|
logger = logging.getLogger(__name__)
|
|
25
25
|
|
verifiers/v1/cli/eval/resume.py
CHANGED
|
@@ -22,7 +22,7 @@ from typing import TypeVar
|
|
|
22
22
|
from pydantic_core import from_json
|
|
23
23
|
|
|
24
24
|
from verifiers.v1.cli.output import CONFIG_FILE, TRACES_FILE, sniff_episode
|
|
25
|
-
from verifiers.v1.configs.eval import EvalConfig
|
|
25
|
+
from verifiers.v1.configs.cli.eval import EvalConfig
|
|
26
26
|
from verifiers.v1.episode import Episode, WireEpisode
|
|
27
27
|
from verifiers.v1.trace import WireTrace
|
|
28
28
|
|
verifiers/v1/cli/eval/runner.py
CHANGED
|
@@ -6,7 +6,7 @@ import logging
|
|
|
6
6
|
import time
|
|
7
7
|
|
|
8
8
|
from verifiers.v1.clients import ModelContext, resolve_client
|
|
9
|
-
from verifiers.v1.configs.eval import EvalConfig
|
|
9
|
+
from verifiers.v1.configs.cli.eval import EvalConfig
|
|
10
10
|
from verifiers.v1.cli.eval import resume
|
|
11
11
|
from verifiers.v1.cli.dashboard import dashboard
|
|
12
12
|
from verifiers.v1.cli.output import (
|
|
@@ -108,7 +108,7 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
|
|
|
108
108
|
from functools import partial
|
|
109
109
|
|
|
110
110
|
from verifiers.v1.utils.logging import setup_logging
|
|
111
|
-
from verifiers.v1.configs.env import pool_serve_kwargs
|
|
111
|
+
from verifiers.v1.configs.cli.env import pool_serve_kwargs
|
|
112
112
|
from verifiers.v1.serve import EnvClient, env_config_data, serve_env
|
|
113
113
|
|
|
114
114
|
legacy = config.is_legacy
|
verifiers/v1/cli/init.py
CHANGED
verifiers/v1/cli/output.py
CHANGED
|
@@ -17,7 +17,7 @@ from pathlib import Path
|
|
|
17
17
|
import tomli_w
|
|
18
18
|
from pydantic import BaseModel, TypeAdapter
|
|
19
19
|
|
|
20
|
-
from verifiers.v1.configs.eval import EvalConfig
|
|
20
|
+
from verifiers.v1.configs.cli.eval import EvalConfig
|
|
21
21
|
from verifiers.v1.episode import Episode, WireEpisode
|
|
22
22
|
from verifiers.v1.trace import Trace
|
|
23
23
|
from verifiers.v1.utils.aio import run_shielded
|
verifiers/v1/cli/replay.py
CHANGED
|
@@ -27,9 +27,10 @@ from verifiers.v1.cli.output import (
|
|
|
27
27
|
save_config,
|
|
28
28
|
write_config,
|
|
29
29
|
)
|
|
30
|
-
from verifiers.v1.configs.replay import ReplayConfig
|
|
30
|
+
from verifiers.v1.configs.cli.replay import ReplayConfig
|
|
31
31
|
from verifiers.v1.state import state_cls
|
|
32
32
|
from verifiers.v1.task import Task, WireTaskData, task_data_cls
|
|
33
|
+
from verifiers.v1.configs.agent import WireAgentConfig
|
|
33
34
|
from verifiers.v1.trace import Trace
|
|
34
35
|
from verifiers.v1.utils.interrupt import install_interrupt
|
|
35
36
|
from verifiers.v1.utils.logging import setup_logging
|
|
@@ -86,7 +87,9 @@ async def run_replay(config: ReplayConfig, source: Path, out: Path) -> list[Trac
|
|
|
86
87
|
# An episode may hold no traces (its env hooks failed before any agent ran);
|
|
87
88
|
# there's nothing to re-score, so it drops out in the flatten. Each kept trace
|
|
88
89
|
# remembers its episode's env identity — the replay output re-stamps it.
|
|
89
|
-
episodes = read_episodes(
|
|
90
|
+
episodes = read_episodes(
|
|
91
|
+
source, Trace[WireTaskData, state_cls(task_cls), WireAgentConfig]
|
|
92
|
+
)
|
|
90
93
|
sourced = [(trace, e.env) for e in episodes for trace in e.traces]
|
|
91
94
|
if config.num_traces is not None:
|
|
92
95
|
sourced = sourced[: config.num_traces]
|
verifiers/v1/cli/serve.py
CHANGED
|
@@ -13,8 +13,8 @@ from verifiers.v1.cli.resolve import (
|
|
|
13
13
|
references_config_file,
|
|
14
14
|
with_positional_taskset,
|
|
15
15
|
)
|
|
16
|
-
from verifiers.v1.configs.serve import ServeConfig
|
|
17
|
-
from verifiers.v1.configs.env import pool_serve_kwargs
|
|
16
|
+
from verifiers.v1.configs.cli.serve import ServeConfig
|
|
17
|
+
from verifiers.v1.configs.cli.env import pool_serve_kwargs
|
|
18
18
|
from verifiers.v1.serve import serve_env
|
|
19
19
|
|
|
20
20
|
USAGE = "usage: uv run serve [<taskset-id>] [--env.id <id>] [--id <env-id> (legacy)] [options] [@ file.toml]"
|
verifiers/v1/cli/validate.py
CHANGED
|
@@ -18,7 +18,7 @@ from verifiers.v1.cli.resolve import (
|
|
|
18
18
|
references_config_file,
|
|
19
19
|
with_positional_taskset,
|
|
20
20
|
)
|
|
21
|
-
from verifiers.v1.configs.validate import ValidateConfig
|
|
21
|
+
from verifiers.v1.configs.cli.validate import ValidateConfig
|
|
22
22
|
from verifiers.v1.decorators import invoke
|
|
23
23
|
from verifiers.v1.utils.compile import resolve_runtime_config
|
|
24
24
|
from verifiers.v1.runtimes import make_runtime
|
verifiers/v1/configs/__init__.py
CHANGED
|
@@ -1,7 +1,4 @@
|
|
|
1
|
-
from
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
from verifiers.v1.configs.validate import ValidateConfig
|
|
6
|
-
|
|
7
|
-
__all__ = ["DebugConfig", "EvalConfig", "InitConfig", "ServeConfig", "ValidateConfig"]
|
|
1
|
+
"""The config tree, separated from the logic it configures: one module per
|
|
2
|
+
domain (`configs.agent`, `configs.harness`, ...) plus the CLI entrypoints' run
|
|
3
|
+
configs under `configs.cli`. Config modules import only types and other configs,
|
|
4
|
+
so any module — the trace record included — can embed them without cycles."""
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""One env agent's config: who plays the seat, and its per-run caps."""
|
|
2
|
+
|
|
3
|
+
from pydantic import SerializeAsAny, model_validator
|
|
4
|
+
from pydantic_config import BaseConfig
|
|
5
|
+
|
|
6
|
+
from verifiers.v1.clients import ClientConfig
|
|
7
|
+
from verifiers.v1.configs.harness import HarnessConfig, WireHarnessConfig
|
|
8
|
+
from verifiers.v1.configs.retries import RetryConfig
|
|
9
|
+
from verifiers.v1.runtimes import RuntimeConfig, SubprocessConfig
|
|
10
|
+
from verifiers.v1.types import SamplingConfig
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class TimeoutConfig(BaseConfig):
|
|
14
|
+
"""Per-agent wall-clock timeouts per rollout stage, in seconds (None = no
|
|
15
|
+
limit); each stage falls back to the task's own `TaskTimeout` when unset. An
|
|
16
|
+
interaction's rollout budget is cumulative across its active harness segments
|
|
17
|
+
and pauses while the caller computes the next user turn."""
|
|
18
|
+
|
|
19
|
+
setup: float | None = None # one shared budget: task setup + provisioning
|
|
20
|
+
rollout: float | None = None
|
|
21
|
+
finalize: float | None = None
|
|
22
|
+
scoring: float | None = None
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class AgentConfig(BaseConfig):
|
|
26
|
+
"""One env agent: who plays it, and its per-run caps. It pins only what
|
|
27
|
+
makes it a different actor; everything unpinned falls back — the model context
|
|
28
|
+
to the run's own, the harness to the taskset's default."""
|
|
29
|
+
|
|
30
|
+
harness: SerializeAsAny[HarnessConfig] | None = None
|
|
31
|
+
"""The agent's program (None = the taskset's default harness)."""
|
|
32
|
+
runtime: RuntimeConfig = SubprocessConfig()
|
|
33
|
+
"""Runtime for the harness program — the policy each run provisions its box
|
|
34
|
+
from; tool servers choose their placement separately."""
|
|
35
|
+
model: str | None = None
|
|
36
|
+
"""Model id (None = the run's model, i.e. the policy under evaluation/training)."""
|
|
37
|
+
client: ClientConfig | None = None
|
|
38
|
+
"""Endpoint override (None = the run's client)."""
|
|
39
|
+
sampling: SamplingConfig | None = None
|
|
40
|
+
"""Sampling override (None = the run's sampling)."""
|
|
41
|
+
timeout: TimeoutConfig = TimeoutConfig()
|
|
42
|
+
retries: RetryConfig = RetryConfig()
|
|
43
|
+
"""Whole-run retries: rerun this agent's rollout while its trace ends with a
|
|
44
|
+
retryable error (never into a borrowed box)."""
|
|
45
|
+
max_turns: int | None = None
|
|
46
|
+
"""Max model turns per run (None = no limit). Framework-enforced (the
|
|
47
|
+
interception server refuses turns past it), so it applies to any harness."""
|
|
48
|
+
max_input_tokens: int | None = None
|
|
49
|
+
max_output_tokens: int | None = None
|
|
50
|
+
max_total_tokens: int | None = None
|
|
51
|
+
"""Token caps per run (None = no limit); framework-enforced between turns."""
|
|
52
|
+
|
|
53
|
+
@model_validator(mode="before")
|
|
54
|
+
@classmethod
|
|
55
|
+
def _resolve_harness(cls, data):
|
|
56
|
+
"""Narrow a pinned `harness` to its concrete config type by `id` (absent
|
|
57
|
+
stays None = the taskset's default). The lazy import keeps class-body
|
|
58
|
+
`AgentConfig()` defaults constructible while this module initializes."""
|
|
59
|
+
if isinstance(data, dict) and data.get("harness") is not None:
|
|
60
|
+
from verifiers.v1.loaders import harness_config_type, narrow_plugin_field
|
|
61
|
+
|
|
62
|
+
narrow_plugin_field(data, "harness", harness_config_type, "bash")
|
|
63
|
+
return data
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class WireAgentConfig(AgentConfig):
|
|
67
|
+
"""Wire form for trace records: parses without resolving the harness plugin,
|
|
68
|
+
so records round-trip anywhere — the knobs stay readable on the extra-allow
|
|
69
|
+
`WireHarnessConfig` (see `WireTaskData`)."""
|
|
70
|
+
|
|
71
|
+
harness: SerializeAsAny[WireHarnessConfig] | None = None
|
|
72
|
+
|
|
73
|
+
@model_validator(mode="before")
|
|
74
|
+
@classmethod
|
|
75
|
+
def _resolve_harness(cls, data):
|
|
76
|
+
"""Override: a record read resolves no plugins."""
|
|
77
|
+
return data
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""The CLI entrypoints' run configs — each `uv run <cmd>`'s parsed tree."""
|
|
2
|
+
|
|
3
|
+
from verifiers.v1.configs.cli.eval import EvalConfig
|
|
4
|
+
from verifiers.v1.configs.cli.debug import DebugConfig
|
|
5
|
+
from verifiers.v1.configs.cli.init import InitConfig
|
|
6
|
+
from verifiers.v1.configs.cli.serve import ServeConfig
|
|
7
|
+
from verifiers.v1.configs.cli.validate import ValidateConfig
|
|
8
|
+
|
|
9
|
+
__all__ = ["DebugConfig", "EvalConfig", "InitConfig", "ServeConfig", "ValidateConfig"]
|
|
@@ -6,9 +6,9 @@ from uuid import uuid4
|
|
|
6
6
|
from pydantic import AliasChoices, Field, SerializeAsAny, model_validator
|
|
7
7
|
from pydantic_config import BaseConfig
|
|
8
8
|
|
|
9
|
-
from verifiers.v1.configs.validate import CheckTimeoutConfig
|
|
9
|
+
from verifiers.v1.configs.cli.validate import CheckTimeoutConfig
|
|
10
10
|
from verifiers.v1.runtimes import DockerConfig, RuntimeConfig
|
|
11
|
-
from verifiers.v1.taskset import TasksetConfig
|
|
11
|
+
from verifiers.v1.configs.taskset import TasksetConfig
|
|
12
12
|
|
|
13
13
|
|
|
14
14
|
class DebugConfig(BaseConfig):
|