verifiers 0.2.2.dev59__py3-none-any.whl → 0.2.2.dev61__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/acp/__init__.py +2 -0
- verifiers/v1/acp/_runner.py +24 -3
- verifiers/v1/agent.py +18 -18
- verifiers/v1/harnesses/__init__.py +3 -0
- verifiers/v1/harnesses/openclaw/__init__.py +6 -0
- verifiers/v1/harnesses/openclaw/harness.py +352 -0
- verifiers/v1/rollout.py +23 -15
- {verifiers-0.2.2.dev59.dist-info → verifiers-0.2.2.dev61.dist-info}/METADATA +1 -1
- {verifiers-0.2.2.dev59.dist-info → verifiers-0.2.2.dev61.dist-info}/RECORD +12 -10
- {verifiers-0.2.2.dev59.dist-info → verifiers-0.2.2.dev61.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev59.dist-info → verifiers-0.2.2.dev61.dist-info}/entry_points.txt +0 -0
- {verifiers-0.2.2.dev59.dist-info → verifiers-0.2.2.dev61.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/acp/__init__.py
CHANGED
|
@@ -33,6 +33,7 @@ class ACP:
|
|
|
33
33
|
mcp_urls: dict[str, str] | None = None,
|
|
34
34
|
system_prompt: str | None = None,
|
|
35
35
|
session_path: str | None = None,
|
|
36
|
+
allow_empty_tool_reply: bool = False,
|
|
36
37
|
) -> ProgramResult:
|
|
37
38
|
if prompt is None:
|
|
38
39
|
raise ValueError("ACP requires a prompt")
|
|
@@ -47,6 +48,7 @@ class ACP:
|
|
|
47
48
|
"mcp_urls": mcp_urls or {},
|
|
48
49
|
"system_prompt": system_prompt or "",
|
|
49
50
|
"session_path": session_path,
|
|
51
|
+
"allow_empty_tool_reply": allow_empty_tool_reply,
|
|
50
52
|
}
|
|
51
53
|
program = await runtime.prepare_uv_script(
|
|
52
54
|
ACP_SOURCE, {**env, "UV_FROZEN": "false"}
|
verifiers/v1/acp/_runner.py
CHANGED
|
@@ -28,6 +28,8 @@ from acp.schema import (
|
|
|
28
28
|
PermissionOption,
|
|
29
29
|
RequestPermissionResponse,
|
|
30
30
|
TextContentBlock,
|
|
31
|
+
ToolCall,
|
|
32
|
+
ToolCallUpdate,
|
|
31
33
|
)
|
|
32
34
|
|
|
33
35
|
|
|
@@ -35,8 +37,16 @@ class VerifiersClient(Client):
|
|
|
35
37
|
def __init__(self) -> None:
|
|
36
38
|
self.visible_reply = ""
|
|
37
39
|
self.message_id: str | None = None
|
|
40
|
+
self.tool_calls: dict[str, str] = {}
|
|
38
41
|
|
|
39
42
|
async def session_update(self, session_id: str, update: Any, **kwargs: Any) -> None:
|
|
43
|
+
if isinstance(update, ToolCall):
|
|
44
|
+
self.tool_calls[update.tool_call_id] = update.status or "pending"
|
|
45
|
+
return
|
|
46
|
+
if isinstance(update, ToolCallUpdate):
|
|
47
|
+
if update.status:
|
|
48
|
+
self.tool_calls[update.tool_call_id] = update.status
|
|
49
|
+
return
|
|
40
50
|
if not isinstance(update, AgentMessageChunk) or not isinstance(
|
|
41
51
|
update.content, TextContentBlock
|
|
42
52
|
):
|
|
@@ -168,13 +178,24 @@ async def run_client(config: dict) -> None:
|
|
|
168
178
|
raise ValueError("ACP prompt has no content")
|
|
169
179
|
client.visible_reply = ""
|
|
170
180
|
client.message_id = None
|
|
181
|
+
client.tool_calls = {}
|
|
171
182
|
try:
|
|
172
|
-
await connection.prompt(session_id=session_id, prompt=prompt)
|
|
183
|
+
response = await connection.prompt(session_id=session_id, prompt=prompt)
|
|
173
184
|
except RequestError as error:
|
|
174
185
|
detail = error.data.get("details") if isinstance(error.data, dict) else None
|
|
175
186
|
raise RuntimeError(detail or str(error)) from error
|
|
176
|
-
|
|
177
|
-
|
|
187
|
+
tool_statuses = list(client.tool_calls.values())
|
|
188
|
+
completed_tool_turn = (
|
|
189
|
+
config["allow_empty_tool_reply"]
|
|
190
|
+
and response.stop_reason == "end_turn"
|
|
191
|
+
and bool(tool_statuses)
|
|
192
|
+
and all(status in ("completed", "failed") for status in tool_statuses)
|
|
193
|
+
)
|
|
194
|
+
if not client.visible_reply.strip() and not completed_tool_turn:
|
|
195
|
+
raise RuntimeError(
|
|
196
|
+
"ACP agent produced no visible reply "
|
|
197
|
+
f"(stop_reason={response.stop_reason}, tool_statuses={tool_statuses})"
|
|
198
|
+
)
|
|
178
199
|
sys.stdout.write(client.visible_reply)
|
|
179
200
|
if session_path and is_new:
|
|
180
201
|
session_path.parent.mkdir(parents=True, exist_ok=True)
|
verifiers/v1/agent.py
CHANGED
|
@@ -25,7 +25,7 @@ from verifiers.v1.harness import Harness
|
|
|
25
25
|
from verifiers.v1.interception import Interception, InterceptionServer
|
|
26
26
|
from verifiers.v1.mcp import SharedToolServer
|
|
27
27
|
from verifiers.v1.retries import backoff, trace_should_retry
|
|
28
|
-
from verifiers.v1.rollout import RolloutRun, _as_messages
|
|
28
|
+
from verifiers.v1.rollout import RolloutRun, RolloutTimeouts, _as_messages
|
|
29
29
|
from verifiers.v1.runtimes import (
|
|
30
30
|
NetworkPolicyConfig,
|
|
31
31
|
Runtime,
|
|
@@ -533,23 +533,23 @@ class Agent:
|
|
|
533
533
|
"harness": self.harness,
|
|
534
534
|
"ctx": self.ctx,
|
|
535
535
|
"runtime_config": runtime_config,
|
|
536
|
-
"
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
agent_timeout, runtime_config, task
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
536
|
+
"timeouts": RolloutTimeouts(
|
|
537
|
+
setup=(
|
|
538
|
+
self.timeout.setup
|
|
539
|
+
if self.timeout.setup is not None
|
|
540
|
+
else task.data.timeout.setup
|
|
541
|
+
),
|
|
542
|
+
agent=cap_remote_agent_timeout(agent_timeout, runtime_config, task),
|
|
543
|
+
finalize=(
|
|
544
|
+
self.timeout.finalize
|
|
545
|
+
if self.timeout.finalize is not None
|
|
546
|
+
else task.data.timeout.finalize
|
|
547
|
+
),
|
|
548
|
+
scoring=(
|
|
549
|
+
self.timeout.scoring
|
|
550
|
+
if self.timeout.scoring is not None
|
|
551
|
+
else task.data.timeout.scoring
|
|
552
|
+
),
|
|
553
553
|
),
|
|
554
554
|
"limits": self.limits,
|
|
555
555
|
"shared_tools": shared_tools,
|
|
@@ -18,6 +18,7 @@ from verifiers.v1.harnesses.mini_swe_agent import (
|
|
|
18
18
|
MiniSWEAgentHarnessConfig,
|
|
19
19
|
)
|
|
20
20
|
from verifiers.v1.harnesses.null import NullHarness, NullHarnessConfig
|
|
21
|
+
from verifiers.v1.harnesses.openclaw import OpenClawHarness, OpenClawHarnessConfig
|
|
21
22
|
from verifiers.v1.harnesses.pi import PiHarness, PiHarnessConfig
|
|
22
23
|
from verifiers.v1.harnesses.pool import PoolHarness, PoolHarnessConfig
|
|
23
24
|
from verifiers.v1.harnesses.rlm import RLMHarness, RLMHarnessConfig
|
|
@@ -40,6 +41,8 @@ __all__ = [
|
|
|
40
41
|
"MiniSWEAgentHarnessConfig",
|
|
41
42
|
"NullHarness",
|
|
42
43
|
"NullHarnessConfig",
|
|
44
|
+
"OpenClawHarness",
|
|
45
|
+
"OpenClawHarnessConfig",
|
|
43
46
|
"PiHarness",
|
|
44
47
|
"PiHarnessConfig",
|
|
45
48
|
"PoolHarness",
|
|
@@ -0,0 +1,352 @@
|
|
|
1
|
+
"""Run OpenClaw's Gateway-backed ACP agent against interception."""
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
import fcntl
|
|
5
|
+
import json
|
|
6
|
+
import logging
|
|
7
|
+
import secrets
|
|
8
|
+
import shlex
|
|
9
|
+
from collections.abc import AsyncIterator
|
|
10
|
+
from contextlib import asynccontextmanager
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from pydantic import Field
|
|
14
|
+
|
|
15
|
+
from verifiers.v1.acp import ACP
|
|
16
|
+
from verifiers.v1.clients import ModelContext
|
|
17
|
+
from verifiers.v1.configs.harness import HarnessConfig
|
|
18
|
+
from verifiers.v1.harness import Harness
|
|
19
|
+
from verifiers.v1.runtimes import DockerRuntime, ProgramResult, Runtime
|
|
20
|
+
from verifiers.v1.task import TaskData
|
|
21
|
+
from verifiers.v1.trace import Trace
|
|
22
|
+
|
|
23
|
+
logger = logging.getLogger(__name__)
|
|
24
|
+
|
|
25
|
+
# OpenClaw and its bundled Node runtime exceed the small /tmp tmpfs in some VMs.
|
|
26
|
+
OPENCLAW_DIR = "/var/tmp/vf-openclaw-{version}"
|
|
27
|
+
OPENCLAW_BIN = f"{OPENCLAW_DIR}/bin/openclaw"
|
|
28
|
+
INSTALL = r"""
|
|
29
|
+
set -e
|
|
30
|
+
command -v curl >/dev/null || (apt-get update -qq && apt-get install -y -qq curl ca-certificates >/dev/null)
|
|
31
|
+
curl -fsSL --proto '=https' --tlsv1.2 https://openclaw.ai/install-cli.sh | bash -s -- --prefix {dir} --version {version} --no-onboard
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
OPENCLAW_ACP = ACP()
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@asynccontextmanager
|
|
38
|
+
async def _gateway_start_lock(runtime: Runtime) -> AsyncIterator[None]:
|
|
39
|
+
if not isinstance(runtime, DockerRuntime) or runtime.network_restricted:
|
|
40
|
+
yield
|
|
41
|
+
return
|
|
42
|
+
# Unrestricted Docker containers share the host network across server workers.
|
|
43
|
+
with Path("/tmp/vf-openclaw-gateway.lock").open("a") as lock: # noqa: ASYNC230
|
|
44
|
+
while True:
|
|
45
|
+
try:
|
|
46
|
+
fcntl.flock(lock, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
47
|
+
break
|
|
48
|
+
except BlockingIOError:
|
|
49
|
+
await asyncio.sleep(0.1)
|
|
50
|
+
try:
|
|
51
|
+
yield
|
|
52
|
+
finally:
|
|
53
|
+
fcntl.flock(lock, fcntl.LOCK_UN)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class OpenClawHarnessConfig(HarnessConfig):
|
|
57
|
+
version: str = Field(default="2026.7.1-2", pattern=r"^[A-Za-z0-9._+-]+$")
|
|
58
|
+
"""OpenClaw release to install, pinned for reproducibility."""
|
|
59
|
+
use_bundled_skill: bool = True
|
|
60
|
+
"""Enable OpenClaw's bundled skill catalog in addition to uploaded harness skills."""
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class OpenClawHarness(Harness[OpenClawHarnessConfig]):
|
|
64
|
+
APPENDS_SYSTEM_PROMPT = True
|
|
65
|
+
SUPPORTS_MCP = True
|
|
66
|
+
SUPPORTS_RESUME = True
|
|
67
|
+
SUPPORTS_SKILLS = True
|
|
68
|
+
|
|
69
|
+
async def setup(self, runtime: Runtime) -> None:
|
|
70
|
+
if not hasattr(self, "_staged_skills_dir"):
|
|
71
|
+
self._staged_skills_dir = (
|
|
72
|
+
f".vf-openclaw/staged-skills-{secrets.token_hex(8)}"
|
|
73
|
+
)
|
|
74
|
+
self._skills_setup_lock = asyncio.Lock()
|
|
75
|
+
if self.config.skills:
|
|
76
|
+
async with self._skills_setup_lock:
|
|
77
|
+
# A complete tree is immutable, so concurrent setups can safely reuse it.
|
|
78
|
+
ready_path = f"{self._staged_skills_dir}/.ready"
|
|
79
|
+
ready = await runtime.run(["test", "-f", ready_path], {})
|
|
80
|
+
if ready.exit_code != 0:
|
|
81
|
+
cleared = await runtime.run(
|
|
82
|
+
["rm", "-rf", self._staged_skills_dir], {}
|
|
83
|
+
)
|
|
84
|
+
if cleared.exit_code != 0:
|
|
85
|
+
raise RuntimeError(
|
|
86
|
+
"failed to clear OpenClaw skills: "
|
|
87
|
+
f"{cleared.stderr.strip()[-500:]}"
|
|
88
|
+
)
|
|
89
|
+
await self.install_skills(runtime, self._staged_skills_dir)
|
|
90
|
+
await runtime.write(ready_path, b"")
|
|
91
|
+
directory = OPENCLAW_DIR.format(version=self.config.version)
|
|
92
|
+
binary = OPENCLAW_BIN.format(version=self.config.version)
|
|
93
|
+
script = INSTALL.replace("{version}", self.config.version).replace(
|
|
94
|
+
"{dir}", directory
|
|
95
|
+
)
|
|
96
|
+
ensure = shlex.quote(f"[ -x {binary} ] || ({script})")
|
|
97
|
+
guarded = (
|
|
98
|
+
f"mkdir -p {directory} && "
|
|
99
|
+
f'"$(command -v flock || command -v lockf)" {directory}/install.lock '
|
|
100
|
+
f"bash -o pipefail -c {ensure}"
|
|
101
|
+
)
|
|
102
|
+
logger.info("openclaw: ensuring OpenClaw %s is installed", self.config.version)
|
|
103
|
+
result = await runtime.run(["sh", "-c", guarded], self.config.resolved_env)
|
|
104
|
+
if result.exit_code != 0:
|
|
105
|
+
detail = (result.stderr or result.stdout).strip()[-500:]
|
|
106
|
+
raise RuntimeError(f"OpenClaw install failed: {detail}")
|
|
107
|
+
await OPENCLAW_ACP.setup(self, runtime)
|
|
108
|
+
|
|
109
|
+
async def launch(
|
|
110
|
+
self,
|
|
111
|
+
ctx: ModelContext,
|
|
112
|
+
trace: Trace,
|
|
113
|
+
runtime: Runtime,
|
|
114
|
+
endpoint: str,
|
|
115
|
+
secret: str,
|
|
116
|
+
mcp_urls: dict[str, str],
|
|
117
|
+
data: TaskData,
|
|
118
|
+
) -> ProgramResult:
|
|
119
|
+
system_prompt, prompt = self.resolve_prompt(data)
|
|
120
|
+
reasoning = ctx.sampling.reasoning_effort not in (
|
|
121
|
+
None,
|
|
122
|
+
"none",
|
|
123
|
+
) or ctx.model.rsplit("/", 1)[-1].startswith(("gpt-5", "o1", "o3", "o4"))
|
|
124
|
+
directory = OPENCLAW_DIR.format(version=self.config.version)
|
|
125
|
+
state_dir = f".vf-openclaw/{trace.id}"
|
|
126
|
+
config_path = f"{state_dir}/openclaw.json"
|
|
127
|
+
skills_dir = f"{state_dir}/skills"
|
|
128
|
+
config = {
|
|
129
|
+
"gateway": {"mode": "local"},
|
|
130
|
+
"agents": {
|
|
131
|
+
"defaults": {
|
|
132
|
+
"workspace": ".",
|
|
133
|
+
"skipBootstrap": True,
|
|
134
|
+
"sandbox": {"mode": "off"},
|
|
135
|
+
"model": {"primary": f"intercept/{ctx.model}"},
|
|
136
|
+
}
|
|
137
|
+
},
|
|
138
|
+
"tools": {
|
|
139
|
+
"profile": "coding",
|
|
140
|
+
"fs": {"workspaceOnly": True},
|
|
141
|
+
"exec": {"host": "gateway", "mode": "full"},
|
|
142
|
+
"deny": self.config.disabled_tools or [],
|
|
143
|
+
},
|
|
144
|
+
"models": {
|
|
145
|
+
"mode": "replace",
|
|
146
|
+
"providers": {
|
|
147
|
+
"intercept": {
|
|
148
|
+
"baseUrl": endpoint,
|
|
149
|
+
"apiKey": "${OPENCLAW_INTERCEPT_KEY}",
|
|
150
|
+
"api": "openai-responses",
|
|
151
|
+
"authHeader": True,
|
|
152
|
+
"models": [
|
|
153
|
+
{
|
|
154
|
+
"id": ctx.model,
|
|
155
|
+
"name": ctx.model,
|
|
156
|
+
"reasoning": reasoning,
|
|
157
|
+
"input": ["text", "image"],
|
|
158
|
+
}
|
|
159
|
+
],
|
|
160
|
+
}
|
|
161
|
+
},
|
|
162
|
+
},
|
|
163
|
+
"mcp": {
|
|
164
|
+
"servers": {
|
|
165
|
+
name: {
|
|
166
|
+
"url": url,
|
|
167
|
+
"transport": "streamable-http",
|
|
168
|
+
"connectionTimeoutMs": 60_000,
|
|
169
|
+
"requestTimeoutMs": 600_000,
|
|
170
|
+
}
|
|
171
|
+
for name, url in mcp_urls.items()
|
|
172
|
+
}
|
|
173
|
+
},
|
|
174
|
+
}
|
|
175
|
+
created = await runtime.run(["mkdir", "-p", state_dir], {})
|
|
176
|
+
if created.exit_code != 0:
|
|
177
|
+
raise RuntimeError(
|
|
178
|
+
f"OpenClaw state directory failed: {created.stderr.strip()[-500:]}"
|
|
179
|
+
)
|
|
180
|
+
if self.config.skills:
|
|
181
|
+
exists = await runtime.run(["test", "-d", skills_dir], {})
|
|
182
|
+
if exists.exit_code != 0:
|
|
183
|
+
copied = await runtime.run(
|
|
184
|
+
["cp", "-R", self._staged_skills_dir, skills_dir], {}
|
|
185
|
+
)
|
|
186
|
+
if copied.exit_code != 0:
|
|
187
|
+
raise RuntimeError(
|
|
188
|
+
f"failed to stage OpenClaw skills: {copied.stderr.strip()[-500:]}"
|
|
189
|
+
)
|
|
190
|
+
config["skills"] = {"load": {"extraDirs": [skills_dir]}}
|
|
191
|
+
if not self.config.use_bundled_skill:
|
|
192
|
+
# OpenClaw treats an empty allowlist as all; a no-match key disables the catalog.
|
|
193
|
+
config.setdefault("skills", {})["allowBundled"] = ["__none__"]
|
|
194
|
+
await runtime.write(config_path, json.dumps(config).encode())
|
|
195
|
+
|
|
196
|
+
binary = OPENCLAW_BIN.format(version=self.config.version)
|
|
197
|
+
node = f"{directory}/tools/node/bin/node"
|
|
198
|
+
allocate_port = shlex.join(
|
|
199
|
+
[
|
|
200
|
+
node,
|
|
201
|
+
"-e",
|
|
202
|
+
(
|
|
203
|
+
'const net=require("node:net");const server=net.createServer();'
|
|
204
|
+
'server.listen(0,"127.0.0.1",()=>{'
|
|
205
|
+
"process.stdout.write(String(server.address().port));server.close();});"
|
|
206
|
+
),
|
|
207
|
+
]
|
|
208
|
+
)
|
|
209
|
+
log_path = f"{state_dir}/gateway.log"
|
|
210
|
+
pid_path = f"{state_dir}/gateway.pid"
|
|
211
|
+
env = {
|
|
212
|
+
**self.config.resolved_env,
|
|
213
|
+
"OPENCLAW_CONFIG_PATH": config_path,
|
|
214
|
+
"OPENCLAW_STATE_DIR": state_dir,
|
|
215
|
+
"OPENCLAW_INTERCEPT_KEY": secret,
|
|
216
|
+
"OPENCLAW_HIDE_BANNER": "1",
|
|
217
|
+
"OPENCLAW_SUPPRESS_NOTES": "1",
|
|
218
|
+
"NO_COLOR": "1",
|
|
219
|
+
}
|
|
220
|
+
async with _gateway_start_lock(runtime):
|
|
221
|
+
allocated = await runtime.run(["sh", "-c", allocate_port], {})
|
|
222
|
+
if allocated.exit_code != 0:
|
|
223
|
+
raise RuntimeError(
|
|
224
|
+
f"OpenClaw port allocation failed: {allocated.stderr.strip()[-500:]}"
|
|
225
|
+
)
|
|
226
|
+
gateway_port = allocated.stdout.strip()
|
|
227
|
+
port_probe = shlex.join(
|
|
228
|
+
[
|
|
229
|
+
node,
|
|
230
|
+
"-e",
|
|
231
|
+
(
|
|
232
|
+
'const net=require("node:net");'
|
|
233
|
+
f'const socket=net.connect({{host:"127.0.0.1",port:{gateway_port}}});'
|
|
234
|
+
'socket.on("connect",()=>socket.end());'
|
|
235
|
+
'socket.on("error",()=>process.exit(1));'
|
|
236
|
+
),
|
|
237
|
+
]
|
|
238
|
+
)
|
|
239
|
+
await runtime.run_background(
|
|
240
|
+
[
|
|
241
|
+
"sh",
|
|
242
|
+
"-c",
|
|
243
|
+
(
|
|
244
|
+
f"echo $$ >{shlex.quote(pid_path)}; "
|
|
245
|
+
f"exec {shlex.quote(binary)} gateway run "
|
|
246
|
+
f"--port {shlex.quote(gateway_port)} --bind loopback "
|
|
247
|
+
f"--auth token --token {shlex.quote(trace.id)} "
|
|
248
|
+
"--allow-unconfigured"
|
|
249
|
+
),
|
|
250
|
+
],
|
|
251
|
+
env,
|
|
252
|
+
log_path,
|
|
253
|
+
)
|
|
254
|
+
bound = await runtime.run(
|
|
255
|
+
[
|
|
256
|
+
"sh",
|
|
257
|
+
"-c",
|
|
258
|
+
(
|
|
259
|
+
"attempt=0; "
|
|
260
|
+
f"until [ -s {shlex.quote(pid_path)} ] && "
|
|
261
|
+
f'kill -0 "$(cat {shlex.quote(pid_path)})" 2>/dev/null && '
|
|
262
|
+
f"{port_probe} >/dev/null 2>&1; do "
|
|
263
|
+
f"if [ -s {shlex.quote(pid_path)} ] && "
|
|
264
|
+
f'! kill -0 "$(cat {shlex.quote(pid_path)})" 2>/dev/null; then '
|
|
265
|
+
f"tail -100 {shlex.quote(log_path)} >&2; exit 1; fi; "
|
|
266
|
+
"attempt=$((attempt + 1)); "
|
|
267
|
+
f'[ "$attempt" -lt 1200 ] || {{ tail -100 {shlex.quote(log_path)} >&2; exit 1; }}; '
|
|
268
|
+
"sleep 0.1; done"
|
|
269
|
+
),
|
|
270
|
+
],
|
|
271
|
+
{},
|
|
272
|
+
)
|
|
273
|
+
if bound.exit_code != 0:
|
|
274
|
+
detail = (bound.stderr or bound.stdout).strip()[-2000:]
|
|
275
|
+
raise RuntimeError(f"OpenClaw gateway failed to bind: {detail}")
|
|
276
|
+
readiness = await runtime.run(
|
|
277
|
+
[
|
|
278
|
+
"sh",
|
|
279
|
+
"-c",
|
|
280
|
+
(
|
|
281
|
+
"attempt=0; "
|
|
282
|
+
f"until [ -s {shlex.quote(pid_path)} ] && "
|
|
283
|
+
f'kill -0 "$(cat {shlex.quote(pid_path)})" 2>/dev/null && '
|
|
284
|
+
f'curl -fsS "http://127.0.0.1:{gateway_port}/healthz" '
|
|
285
|
+
">/dev/null 2>&1; do "
|
|
286
|
+
f"if [ -s {shlex.quote(pid_path)} ] && "
|
|
287
|
+
f'! kill -0 "$(cat {shlex.quote(pid_path)})" 2>/dev/null; then '
|
|
288
|
+
f"tail -100 {shlex.quote(log_path)} >&2; exit 1; fi; "
|
|
289
|
+
"attempt=$((attempt + 1)); "
|
|
290
|
+
f'[ "$attempt" -lt 120 ] || {{ tail -100 {shlex.quote(log_path)} >&2; exit 1; }}; '
|
|
291
|
+
"sleep 1; done"
|
|
292
|
+
),
|
|
293
|
+
],
|
|
294
|
+
{},
|
|
295
|
+
)
|
|
296
|
+
if readiness.exit_code != 0:
|
|
297
|
+
detail = (readiness.stderr or readiness.stdout).strip()[-2000:]
|
|
298
|
+
raise RuntimeError(f"OpenClaw gateway failed to start: {detail}")
|
|
299
|
+
command = [
|
|
300
|
+
"sh",
|
|
301
|
+
"-c",
|
|
302
|
+
(
|
|
303
|
+
"set -eu; "
|
|
304
|
+
f"gateway_pid=$(cat {shlex.quote(pid_path)}); "
|
|
305
|
+
f'trap \'kill "$gateway_pid" 2>/dev/null || true; '
|
|
306
|
+
'attempt=0; while kill -0 "$gateway_pid" 2>/dev/null && '
|
|
307
|
+
' [ "$attempt" -lt 50 ]; do attempt=$((attempt + 1)); sleep 0.1; done; '
|
|
308
|
+
'kill -9 "$gateway_pid" 2>/dev/null || true; '
|
|
309
|
+
'while kill -0 "$gateway_pid" 2>/dev/null; do sleep 0.1; done; '
|
|
310
|
+
f"rm -f {shlex.quote(pid_path)}' EXIT; "
|
|
311
|
+
f'{shlex.quote(binary)} acp --url "ws://127.0.0.1:{gateway_port}" '
|
|
312
|
+
f"--token {shlex.quote(trace.id)} --no-prefix-cwd"
|
|
313
|
+
),
|
|
314
|
+
]
|
|
315
|
+
# OpenClaw rejects ACP per-session MCP declarations; the isolated Gateway
|
|
316
|
+
# config above owns the equivalent task-scoped server definitions.
|
|
317
|
+
return await OPENCLAW_ACP.run(
|
|
318
|
+
runtime,
|
|
319
|
+
env,
|
|
320
|
+
command,
|
|
321
|
+
prompt,
|
|
322
|
+
mcp_urls={},
|
|
323
|
+
system_prompt=system_prompt,
|
|
324
|
+
session_path=f"{state_dir}/acp-session",
|
|
325
|
+
# OpenClaw can end after its final tool completes without a text message.
|
|
326
|
+
allow_empty_tool_reply=True,
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
async def cleanup(self, trace: Trace, runtime: Runtime) -> None:
|
|
330
|
+
state_dir = f".vf-openclaw/{trace.id}"
|
|
331
|
+
pid_path = f"{state_dir}/gateway.pid"
|
|
332
|
+
result = await runtime.run(
|
|
333
|
+
[
|
|
334
|
+
"sh",
|
|
335
|
+
"-c",
|
|
336
|
+
(
|
|
337
|
+
f"if [ -s {shlex.quote(pid_path)} ]; then "
|
|
338
|
+
f"gateway_pid=$(cat {shlex.quote(pid_path)}); "
|
|
339
|
+
'kill "$gateway_pid" 2>/dev/null || true; '
|
|
340
|
+
'attempt=0; while kill -0 "$gateway_pid" 2>/dev/null && '
|
|
341
|
+
' [ "$attempt" -lt 50 ]; do attempt=$((attempt + 1)); sleep 0.1; done; '
|
|
342
|
+
'kill -9 "$gateway_pid" 2>/dev/null || true; '
|
|
343
|
+
'while kill -0 "$gateway_pid" 2>/dev/null; do sleep 0.1; done; fi; '
|
|
344
|
+
f"rm -rf {shlex.quote(state_dir)}"
|
|
345
|
+
),
|
|
346
|
+
],
|
|
347
|
+
{},
|
|
348
|
+
)
|
|
349
|
+
if result.exit_code != 0:
|
|
350
|
+
raise RuntimeError(
|
|
351
|
+
f"failed to clean up OpenClaw state: {result.stderr.strip()[-500:]}"
|
|
352
|
+
)
|
verifiers/v1/rollout.py
CHANGED
|
@@ -23,6 +23,7 @@ import logging
|
|
|
23
23
|
import time
|
|
24
24
|
from collections.abc import AsyncIterator, Callable
|
|
25
25
|
from contextlib import AsyncExitStack, asynccontextmanager
|
|
26
|
+
from dataclasses import dataclass
|
|
26
27
|
|
|
27
28
|
from verifiers.v1.clients import ModelContext
|
|
28
29
|
from verifiers.v1.configs.agent import AgentConfig
|
|
@@ -57,6 +58,19 @@ from verifiers.v1.types import Messages
|
|
|
57
58
|
logger = logging.getLogger(__name__)
|
|
58
59
|
|
|
59
60
|
|
|
61
|
+
@dataclass(frozen=True)
|
|
62
|
+
class RolloutTimeouts:
|
|
63
|
+
"""Per-stage rollout timeouts in seconds (None = unlimited), each bounding one
|
|
64
|
+
lifecycle stage: `setup` covers task + harness setup in `open()`, `agent` is the
|
|
65
|
+
cumulative budget the run's segments draw down across `step()`s, `finalize` and
|
|
66
|
+
`scoring` bound their `close()` stages."""
|
|
67
|
+
|
|
68
|
+
setup: float | None = None
|
|
69
|
+
agent: float | None = None
|
|
70
|
+
finalize: float | None = None
|
|
71
|
+
scoring: float | None = None
|
|
72
|
+
|
|
73
|
+
|
|
60
74
|
def _as_messages(raw: Messages) -> Messages:
|
|
61
75
|
"""A turn's messages may arrive typed or as wire dicts (env code naturally
|
|
62
76
|
writes `{"role": "user", ...}`); the trace speaks typed, so normalize here."""
|
|
@@ -115,10 +129,7 @@ class RolloutRun:
|
|
|
115
129
|
runtime_config: RuntimeConfig,
|
|
116
130
|
wire_data: TaskData | None = None,
|
|
117
131
|
has_user: bool = False,
|
|
118
|
-
|
|
119
|
-
agent_timeout: float | None = None,
|
|
120
|
-
finalize_timeout: float | None = None,
|
|
121
|
-
scoring_timeout: float | None = None,
|
|
132
|
+
timeouts: RolloutTimeouts | None = None,
|
|
122
133
|
limits: RolloutLimits | None = None,
|
|
123
134
|
shared_tools: dict[str, SharedToolServer] | None = None,
|
|
124
135
|
interception: Interception | None = None,
|
|
@@ -130,11 +141,8 @@ class RolloutRun:
|
|
|
130
141
|
self.ctx = ctx
|
|
131
142
|
self.runtime_config = runtime_config
|
|
132
143
|
self._has_user = has_user
|
|
133
|
-
self.
|
|
134
|
-
self.
|
|
135
|
-
self._agent_time_remaining = agent_timeout
|
|
136
|
-
self._finalize_timeout = finalize_timeout
|
|
137
|
-
self._scoring_timeout = scoring_timeout
|
|
144
|
+
self._timeouts = timeouts or RolloutTimeouts()
|
|
145
|
+
self._agent_time_remaining = self._timeouts.agent
|
|
138
146
|
self._shared_tools = shared_tools or {}
|
|
139
147
|
self._interception = interception
|
|
140
148
|
self.runtime = runtime
|
|
@@ -164,7 +172,7 @@ class RolloutRun:
|
|
|
164
172
|
self.deadline_at: float | None = None
|
|
165
173
|
"""The active harness segment's absolute deadline (event-loop clock), or
|
|
166
174
|
None between segments / when unbounded. An interaction spends one cumulative
|
|
167
|
-
`
|
|
175
|
+
`timeouts.agent` budget only while its own segments run, so time awaiting
|
|
168
176
|
the caller (including another interleaved agent) cannot starve it."""
|
|
169
177
|
|
|
170
178
|
@property
|
|
@@ -244,8 +252,8 @@ class RolloutRun:
|
|
|
244
252
|
# Task setup and harness provisioning share one setup-stage deadline.
|
|
245
253
|
setup_deadline = (
|
|
246
254
|
None
|
|
247
|
-
if self.
|
|
248
|
-
else asyncio.get_running_loop().time() + self.
|
|
255
|
+
if self._timeouts.setup is None
|
|
256
|
+
else asyncio.get_running_loop().time() + self._timeouts.setup
|
|
249
257
|
)
|
|
250
258
|
async with (
|
|
251
259
|
boundary(TaskError, "task setup"),
|
|
@@ -340,7 +348,7 @@ class RolloutRun:
|
|
|
340
348
|
self.fail(
|
|
341
349
|
HarnessError(
|
|
342
350
|
f"agent timeout: rollout exceeded its "
|
|
343
|
-
f"{self.
|
|
351
|
+
f"{self._timeouts.agent:g}s budget"
|
|
344
352
|
)
|
|
345
353
|
)
|
|
346
354
|
else:
|
|
@@ -406,7 +414,7 @@ class RolloutRun:
|
|
|
406
414
|
invoke(
|
|
407
415
|
self.task.finalize, {"trace": trace, "runtime": runtime}
|
|
408
416
|
),
|
|
409
|
-
self.
|
|
417
|
+
self._timeouts.finalize,
|
|
410
418
|
)
|
|
411
419
|
now = time.time()
|
|
412
420
|
trace.timing.finalize.end = now
|
|
@@ -418,7 +426,7 @@ class RolloutRun:
|
|
|
418
426
|
self.task.score(trace, runtime),
|
|
419
427
|
self.harness.score(trace, runtime),
|
|
420
428
|
),
|
|
421
|
-
self.
|
|
429
|
+
self._timeouts.scoring,
|
|
422
430
|
)
|
|
423
431
|
trace.timing.scoring.end = time.time()
|
|
424
432
|
except Exception as e: # noqa: BLE001 - finalize boundary records every rollout failure
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.dev61
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -168,7 +168,7 @@ verifiers/utils/tool_utils.py,sha256=gInWZODWQUZN4TyEMuuITyOUm9qlHJxtcvr1zZl_zJs
|
|
|
168
168
|
verifiers/utils/usage_utils.py,sha256=Mw6WC78r6EsLyRFdP9ICveqptkjizbSkAC86ZH0geVU,3918
|
|
169
169
|
verifiers/utils/version_utils.py,sha256=39gV8-mNF9GspAgKloLo8oy7_MVluMrwGPrntQv2hx4,2656
|
|
170
170
|
verifiers/v1/__init__.py,sha256=Im5qTZy9mH6PRTUi-1V6W7EWKZzJyTWLGKMYALbWz_w,7467
|
|
171
|
-
verifiers/v1/agent.py,sha256=
|
|
171
|
+
verifiers/v1/agent.py,sha256=ASbe6IHAYcwivzQkHunX3cGu-o10Gez8tlVhO-pLYyE,32134
|
|
172
172
|
verifiers/v1/artifacts.py,sha256=5eKOjDSUWwj7GrhxILMHdz2ngEONWMdcTGlBcuGgrLU,5834
|
|
173
173
|
verifiers/v1/decorators.py,sha256=9CcCwgZlRjVGPqT5G1CFJUHP4OIMuBGkUJZkcI6dnbk,3865
|
|
174
174
|
verifiers/v1/env.py,sha256=TGCq_rPp_VdyGAKpHuscSDyIxIGnim2N-dpKkG-J5p0,18175
|
|
@@ -181,7 +181,7 @@ verifiers/v1/legacy.py,sha256=ISk3Bix9Hy5EjZtyV9QO0LNyDNDklBHVjFokOrw_3U8,22628
|
|
|
181
181
|
verifiers/v1/loaders.py,sha256=NSIBdNU3JVeyddL-dM4dJt6khCHyjCgzJuFtLrgiwD0,9989
|
|
182
182
|
verifiers/v1/push.py,sha256=MWZmvyj0iyzNIO0xRhlkCXtFF1plq6EJi8rWlvrFtvI,11363
|
|
183
183
|
verifiers/v1/retries.py,sha256=quUGVruUkZx-9WQcFiocMfL_M056voU8gwDqsuLR3f4,5412
|
|
184
|
-
verifiers/v1/rollout.py,sha256=
|
|
184
|
+
verifiers/v1/rollout.py,sha256=s8ji1W19FHZOn-Wn3_fjw-b1hhmi0EqZ7dZVntYK5pM,20732
|
|
185
185
|
verifiers/v1/scoring.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
|
|
186
186
|
verifiers/v1/session.py,sha256=J6ZPuvsg6vMTJAs9dMUMtfdjCuvp6Hom877DVdKFKlY,6912
|
|
187
187
|
verifiers/v1/state.py,sha256=pcpN2V6tX7rIO5XdtdUYay1j0ktEwrMtq7w1ApfsPdE,675
|
|
@@ -189,8 +189,8 @@ verifiers/v1/task.py,sha256=osuZNjGeNj5NC9TYxxzRQUPnt8D5aNw1OuS45eVWugI,9057
|
|
|
189
189
|
verifiers/v1/taskset.py,sha256=s9T7v3REU1EKVCu2zORMA3gWinRK6LJk4GUvoKhAmEQ,4337
|
|
190
190
|
verifiers/v1/trace.py,sha256=_bGHOrE6fRLGm0JYc0PrrrO3lyr-qqOF5Y1bHfViwR0,19347
|
|
191
191
|
verifiers/v1/types.py,sha256=1PxamJspmoTc3OlFZAwH6o_2i_Tg9ZNvHxttIHWIxq8,8948
|
|
192
|
-
verifiers/v1/acp/__init__.py,sha256=
|
|
193
|
-
verifiers/v1/acp/_runner.py,sha256=
|
|
192
|
+
verifiers/v1/acp/__init__.py,sha256=9SYCFtzGUM_wH98ldpVLcFhoq2M999jbG0tU5ODY70U,2297
|
|
193
|
+
verifiers/v1/acp/_runner.py,sha256=BcYaNZHuawzCgxOVdhiF6PY_B1HMxIA1MwHy0mOzhSk,7740
|
|
194
194
|
verifiers/v1/cli/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
195
195
|
verifiers/v1/cli/debug.py,sha256=-jeEV_Q5tV8SOkWc1GdPZcXZWON_jU5gjeO4YoZm79o,11482
|
|
196
196
|
verifiers/v1/cli/gepa.py,sha256=hLIbS_4jBx0vvy4PXoePR5riFXcNvzQCMixZ-g2OR_U,4026
|
|
@@ -252,7 +252,7 @@ verifiers/v1/gepa/config.py,sha256=gFn8Re4GoeHJs2PZ-iw4jCEZ72AiM1lNgwaA0uvJlYw,4
|
|
|
252
252
|
verifiers/v1/gepa/dataset.py,sha256=N2QYqUljHVuAdCbUuydEIlarvAWa6LFdGKu9qsOvjro,2186
|
|
253
253
|
verifiers/v1/gepa/reflection.py,sha256=ptHhx0lcDLcWhWhXuslh9T8gPVNpnX5XLPtXoneW1yA,1054
|
|
254
254
|
verifiers/v1/gepa/runner.py,sha256=-LZEt0lod6NKfyxg2dvOXZjfzTv3AWwendexd1De26c,6260
|
|
255
|
-
verifiers/v1/harnesses/__init__.py,sha256=
|
|
255
|
+
verifiers/v1/harnesses/__init__.py,sha256=cLyHRiH0E63WcVW9dBqR4OEqPSfPR4wQmB6gunghV4Q,1759
|
|
256
256
|
verifiers/v1/harnesses/bash/__init__.py,sha256=IAV4eMMqDHSl-9ANb0Q3kxmBZ22SeHZFBM448X4gQio,140
|
|
257
257
|
verifiers/v1/harnesses/bash/harness.py,sha256=ZtjhsANuK3vXkf9mRCi8-sBMyPebJMBNHa58vTd2Fm8,5284
|
|
258
258
|
verifiers/v1/harnesses/bash/program.py,sha256=hGv3k9jwoabL045AGiZWiX76Cca3Juk58AaMYeJTEGo,15334
|
|
@@ -274,6 +274,8 @@ verifiers/v1/harnesses/mini_swe_agent/program.py,sha256=XPjDEsbJACB34tkUX9dJdIzi
|
|
|
274
274
|
verifiers/v1/harnesses/null/__init__.py,sha256=XDeKOPeoXUQi9US9_Z3_yH9YaZ4EiOMKJmUtNNdtD7s,127
|
|
275
275
|
verifiers/v1/harnesses/null/harness.py,sha256=5qKWiIRfT9uRwCMAPxIJwilfUWpibixymA6NNpDcCIc,2531
|
|
276
276
|
verifiers/v1/harnesses/null/program.py,sha256=RR3hZWPtZ3SnRcBRw8z7PaixLR-vYiI9J640YuXxcXs,8441
|
|
277
|
+
verifiers/v1/harnesses/openclaw/__init__.py,sha256=kNyfPB1sGwEf4K3mTfKN0VEyxun3SGvNpvArVKvH-x8,160
|
|
278
|
+
verifiers/v1/harnesses/openclaw/harness.py,sha256=3fLpwSghM5JPdMjZ7aQQaOOBc_1RdsREpTfatdQXy3o,14616
|
|
277
279
|
verifiers/v1/harnesses/pi/__init__.py,sha256=1mSnzOf-RdEchdeg6IHtcdobG5KIpAhuTdUIgsCIn74,117
|
|
278
280
|
verifiers/v1/harnesses/pi/harness.py,sha256=_O-pyg2RM23-RzQosKHKf8uG8KflS4OrWSsEiBCWQpo,8675
|
|
279
281
|
verifiers/v1/harnesses/pool/__init__.py,sha256=HTYsNiWGdiNEsSHj4xhPmshJSNb7SHjQ7d4yZQ4p3HM,127
|
|
@@ -336,8 +338,8 @@ verifiers/v1/utils/logging.py,sha256=OcMHA6NsYux3oIzjPuI95rDWmFBHNcHDjZeNIhXTX-Y
|
|
|
336
338
|
verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
|
|
337
339
|
verifiers/v1/utils/sampling.py,sha256=52TJ5-Hmpvp4YSgKNR-LmAPVwm4RDcH-o2-BnbVTmYQ,965
|
|
338
340
|
verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
|
|
339
|
-
verifiers-0.2.2.
|
|
340
|
-
verifiers-0.2.2.
|
|
341
|
-
verifiers-0.2.2.
|
|
342
|
-
verifiers-0.2.2.
|
|
343
|
-
verifiers-0.2.2.
|
|
341
|
+
verifiers-0.2.2.dev61.dist-info/METADATA,sha256=vuxmOQykOIiG0GzqOXC2Xm7lC8DGx5EqMqFcjpgt1As,4545
|
|
342
|
+
verifiers-0.2.2.dev61.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
343
|
+
verifiers-0.2.2.dev61.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
|
|
344
|
+
verifiers-0.2.2.dev61.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
345
|
+
verifiers-0.2.2.dev61.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|