verifiers 0.2.2.dev96__py3-none-any.whl → 0.2.2.dev112__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/acp/__init__.py +6 -2
- verifiers/v1/clients/train.py +6 -2
- verifiers/v1/configs/client.py +3 -2
- verifiers/v1/dialects/chat.py +5 -1
- verifiers/v1/envs/agentic_judge/env.py +5 -3
- verifiers/v1/envs/user_sim/env.py +2 -1
- verifiers/v1/gepa/runner.py +1 -1
- verifiers/v1/harnesses/claude_code/harness.py +31 -15
- verifiers/v1/harnesses/codex/harness.py +33 -19
- verifiers/v1/harnesses/kimi_code/harness.py +3 -1
- verifiers/v1/harnesses/mini_swe_agent/harness.py +3 -1
- verifiers/v1/harnesses/openclaw/harness.py +2 -4
- verifiers/v1/harnesses/pi/harness.py +7 -3
- verifiers/v1/harnesses/pool/harness.py +1 -1
- verifiers/v1/harnesses/rlm/harness.py +1 -1
- verifiers/v1/harnesses/terminus_2/harness.py +3 -1
- verifiers/v1/interception/server.py +21 -2
- verifiers/v1/mcp/launch.py +108 -41
- verifiers/v1/mcp/server.py +3 -2
- verifiers/v1/mcp/toolset.py +1 -1
- verifiers/v1/rollout.py +4 -13
- verifiers/v1/runtimes/base.py +9 -0
- verifiers/v1/runtimes/limiters.py +9 -4
- verifiers/v1/session.py +6 -3
- verifiers/v1/tasksets/__init__.py +3 -0
- verifiers/v1/tasksets/nemo_gym/__init__.py +17 -0
- verifiers/v1/tasksets/nemo_gym/response.py +110 -0
- verifiers/v1/tasksets/nemo_gym/server.py +42 -0
- verifiers/v1/tasksets/nemo_gym/taskset.py +170 -0
- verifiers/v1/tasksets/nemo_gym/toolset.py +107 -0
- {verifiers-0.2.2.dev96.dist-info → verifiers-0.2.2.dev112.dist-info}/METADATA +4 -2
- {verifiers-0.2.2.dev96.dist-info → verifiers-0.2.2.dev112.dist-info}/RECORD +35 -30
- {verifiers-0.2.2.dev96.dist-info → verifiers-0.2.2.dev112.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev96.dist-info → verifiers-0.2.2.dev112.dist-info}/entry_points.txt +0 -0
- {verifiers-0.2.2.dev96.dist-info → verifiers-0.2.2.dev112.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/acp/__init__.py
CHANGED
|
@@ -121,7 +121,9 @@ class ACP:
|
|
|
121
121
|
"allow_empty_tool_reply": allow_empty_tool_reply,
|
|
122
122
|
}
|
|
123
123
|
program = await runtime.prepare_uv_script(
|
|
124
|
-
ACP_SOURCE,
|
|
124
|
+
ACP_SOURCE,
|
|
125
|
+
{**env, "UV_FROZEN": "false"},
|
|
126
|
+
activate=False,
|
|
125
127
|
)
|
|
126
128
|
directory = f".vf-acp-{secrets.token_hex(8)}"
|
|
127
129
|
created = await runtime.run(["mkdir", "-m", "700", directory], {})
|
|
@@ -198,7 +200,9 @@ class ACPHarnessSession(HarnessSession):
|
|
|
198
200
|
async def _start(self) -> None:
|
|
199
201
|
self._stderr_tail.clear()
|
|
200
202
|
program = await self.runtime.prepare_uv_script(
|
|
201
|
-
ACP_SOURCE,
|
|
203
|
+
ACP_SOURCE,
|
|
204
|
+
{**self.env, "UV_FROZEN": "false"},
|
|
205
|
+
activate=False,
|
|
202
206
|
)
|
|
203
207
|
process = await self.runtime.open_process([*program, "stream"], self.env)
|
|
204
208
|
self._process = process
|
verifiers/v1/clients/train.py
CHANGED
|
@@ -12,6 +12,7 @@ from typing import Any, ClassVar, TypeVar
|
|
|
12
12
|
from openai import OpenAIError
|
|
13
13
|
from renderers import OverlongPromptError as RendererOverlongPromptError
|
|
14
14
|
from renderers import RenderedTokens, Renderer, RendererConfig
|
|
15
|
+
from renderers.base import ToolCallParseStatus
|
|
15
16
|
|
|
16
17
|
from verifiers.v1.clients.base import build_async_openai
|
|
17
18
|
from verifiers.v1.clients.client import SESSION_ID_HEADER, Client
|
|
@@ -115,6 +116,8 @@ def response_from_generate(
|
|
|
115
116
|
)
|
|
116
117
|
for i, tc in enumerate(result.get("tool_calls") or [])
|
|
117
118
|
if getattr(tc, "name", None)
|
|
119
|
+
# TODO: we need a better way for renderers to expose this
|
|
120
|
+
and getattr(tc, "status", None) != ToolCallParseStatus.UNKNOWN_TOOL
|
|
118
121
|
] or None
|
|
119
122
|
prompt_ids = result.get("prompt_ids") or []
|
|
120
123
|
completion_ids = result.get("completion_ids") or []
|
|
@@ -296,8 +299,9 @@ class ElasticRendererPool:
|
|
|
296
299
|
class TrainClient(Client):
|
|
297
300
|
"""Renders prompts to token ids and calls a vLLM `/inference/v1/generate` engine.
|
|
298
301
|
|
|
299
|
-
|
|
300
|
-
|
|
302
|
+
Owned by the interception server and shared by the rollouts it multiplexes: they reuse
|
|
303
|
+
its engine connection pool, and each turn takes a slot on the shared
|
|
304
|
+
`ElasticRendererPool`."""
|
|
301
305
|
|
|
302
306
|
def __init__(self, config: TrainClientConfig) -> None:
|
|
303
307
|
self.config = config
|
verifiers/v1/configs/client.py
CHANGED
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
"""Client configs: describe an OpenAI-compatible endpoint.
|
|
2
2
|
|
|
3
3
|
A `BaseClientConfig` is an OpenAI-compatible endpoint (base_url + API-key env var
|
|
4
|
-
+ extra headers); `clients.resolve_client` turns one into a live `Client
|
|
5
|
-
|
|
4
|
+
+ extra headers); `clients.resolve_client` turns one into a live `Client` — the
|
|
5
|
+
interception server builds one per distinct config and shares it across the rollouts
|
|
6
|
+
it multiplexes. The default Prime endpoint, API key, and team fall back to
|
|
6
7
|
the active Prime CLI config, so direct `uv run eval` calls behave like `prime eval`.
|
|
7
8
|
Both the eval entrypoint (its model client) and in-env LLM calls (e.g. a judge reward)
|
|
8
9
|
build clients from these. `ClientConfig` is the CLI-selectable discriminated union
|
verifiers/v1/dialects/chat.py
CHANGED
|
@@ -141,7 +141,11 @@ def _content_to_wire(content):
|
|
|
141
141
|
|
|
142
142
|
def message_to_wire(message: Message) -> dict:
|
|
143
143
|
if message.role == "assistant":
|
|
144
|
-
|
|
144
|
+
# Strict providers reject `content: null` without tool calls.
|
|
145
|
+
content = message.content
|
|
146
|
+
if content is None and not message.tool_calls:
|
|
147
|
+
content = ""
|
|
148
|
+
wire: dict = {"role": "assistant", "content": content}
|
|
145
149
|
if message.provider_state:
|
|
146
150
|
wire["reasoning_details"] = message.provider_state
|
|
147
151
|
elif message.reasoning_content is not None:
|
|
@@ -309,8 +309,6 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
|
|
|
309
309
|
|
|
310
310
|
async def finalize(self, task: vf.Task, episode: vf.Episode) -> None:
|
|
311
311
|
by_agent = {t.agent.name: t for t in episode.traces}
|
|
312
|
-
if "judge" not in by_agent:
|
|
313
|
-
return
|
|
314
312
|
solution, verdict = by_agent["solver"], by_agent["judge"]
|
|
315
313
|
verdicts = RubricVerdicts.model_validate(verdict.info.get("verdict")).verdicts
|
|
316
314
|
criteria = self.config.task.criteria()
|
|
@@ -332,6 +330,10 @@ class SharedAgenticJudgeEnv(AgenticJudgeEnv):
|
|
|
332
330
|
async def run(self, task: vf.Task, agents: vf.Agents) -> None:
|
|
333
331
|
async with agents.solver.provision(task) as box:
|
|
334
332
|
solution = await agents.solver.run(task, runtime=box)
|
|
333
|
+
if not solution.ok:
|
|
334
|
+
raise RuntimeError(
|
|
335
|
+
"the solver's rollout failed, so the judge never ran"
|
|
336
|
+
)
|
|
335
337
|
judge_task = JudgeTask.from_trace(solution, self.config.task)
|
|
336
338
|
await agents.judge.run(judge_task, runtime=box)
|
|
337
339
|
|
|
@@ -342,7 +344,7 @@ class IsolatedAgenticJudgeEnv(AgenticJudgeEnv):
|
|
|
342
344
|
async def run(self, task: vf.Task, agents: vf.Agents) -> None:
|
|
343
345
|
solution = await agents.solver.run(task)
|
|
344
346
|
if not solution.ok:
|
|
345
|
-
|
|
347
|
+
raise RuntimeError("the solver's rollout failed, so the judge never ran")
|
|
346
348
|
await agents.judge.run(
|
|
347
349
|
JudgeTask.from_trace(solution, self.config.task, share_runtime=False)
|
|
348
350
|
)
|
|
@@ -21,13 +21,14 @@ from pydantic import Field
|
|
|
21
21
|
|
|
22
22
|
import verifiers.v1 as vf
|
|
23
23
|
|
|
24
|
-
PERSONA = """You are role-playing a USER talking to an AI assistant. This is your situation
|
|
24
|
+
PERSONA = """You are role-playing a USER talking to an AI assistant. This is your situation — what you want the ASSISTANT to do for you:
|
|
25
25
|
|
|
26
26
|
{scenario}
|
|
27
27
|
|
|
28
28
|
Rules:
|
|
29
29
|
- Open the conversation with your request, in your own words.
|
|
30
30
|
- Stay in character: short, natural user messages; never act as the assistant.
|
|
31
|
+
- The assistant does the work, not you: never perform the task yourself — any tool call, answer, or output format it asks for must come from the assistant; ask for it instead.
|
|
31
32
|
- Reveal details only when asked, as a real user would.
|
|
32
33
|
- When the assistant has fully met your goal — or you are convinced it cannot — reply with exactly {done} and nothing else."""
|
|
33
34
|
|
verifiers/v1/gepa/runner.py
CHANGED
|
@@ -73,7 +73,7 @@ def run_gepa(env: Env, config: GEPAConfig) -> GEPAResult:
|
|
|
73
73
|
await append_episode(run_dir, episode, write_lock)
|
|
74
74
|
|
|
75
75
|
try:
|
|
76
|
-
# The endpoint stays config:
|
|
76
|
+
# The endpoint stays config: the interception server builds the live client.
|
|
77
77
|
ctx = ModelContext(
|
|
78
78
|
client=config.client, model=config.model, sampling=config.sampling
|
|
79
79
|
)
|
|
@@ -2,6 +2,8 @@
|
|
|
2
2
|
|
|
3
3
|
import shlex
|
|
4
4
|
|
|
5
|
+
from pydantic import Field
|
|
6
|
+
|
|
5
7
|
from verifiers.v1.acp import ACP
|
|
6
8
|
from verifiers.v1.clients import ModelContext
|
|
7
9
|
from verifiers.v1.configs.harness import HarnessConfig
|
|
@@ -11,31 +13,30 @@ from verifiers.v1.runtimes import ProgramResult, Runtime
|
|
|
11
13
|
from verifiers.v1.task import TaskData
|
|
12
14
|
from verifiers.v1.trace import Trace
|
|
13
15
|
|
|
14
|
-
CLAUDE_ACP_DIR = "/var/tmp/vf-claude-agent-acp"
|
|
16
|
+
CLAUDE_ACP_DIR = "/var/tmp/vf-claude-agent-acp-{version}-{acp_version}"
|
|
15
17
|
PACKAGES_DIR = f"{CLAUDE_ACP_DIR}/packages"
|
|
16
|
-
ACP_VERSION = "0.
|
|
18
|
+
ACP_VERSION = "0.65.0"
|
|
19
|
+
CLAUDE_BIN = f"{PACKAGES_DIR}/node_modules/.bin/claude"
|
|
17
20
|
ACP_BIN = f"{PACKAGES_DIR}/node_modules/.bin/claude-agent-acp"
|
|
18
|
-
ACP_COMMAND = [f"{NODE_BIN_DIR}/node", ACP_BIN]
|
|
19
21
|
CLAUDE_CONFIG_ROOT = ".vf-claude"
|
|
20
22
|
SKILLS_DIR = ".claude/skills"
|
|
21
23
|
ACP_INSTALL = r"""
|
|
22
24
|
set -e
|
|
23
25
|
export PATH="/var/tmp/vf-node/bin:$PATH"
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
exit 0
|
|
27
|
-
fi
|
|
28
|
-
npm install --prefix /var/tmp/vf-claude-agent-acp/packages --ignore-scripts --no-audit --no-fund \
|
|
26
|
+
rm -f {ready}
|
|
27
|
+
npm install --prefix {packages} --no-audit --no-fund \
|
|
29
28
|
--omit=dev \
|
|
29
|
+
"@anthropic-ai/claude-code@$VF_CLAUDE_CODE_VERSION" \
|
|
30
30
|
"@agentclientprotocol/claude-agent-acp@$VF_CLAUDE_ACP_VERSION" >/dev/null
|
|
31
|
-
|
|
31
|
+
touch {ready}
|
|
32
32
|
"""
|
|
33
33
|
|
|
34
34
|
CLAUDE_ACP = ACP()
|
|
35
35
|
|
|
36
36
|
|
|
37
37
|
class ClaudeCodeHarnessConfig(HarnessConfig):
|
|
38
|
-
|
|
38
|
+
version: str = Field(default="2.1.223", pattern=r"^[A-Za-z0-9._+-]+$")
|
|
39
|
+
"""Claude Code release to install, pinned for reproducibility."""
|
|
39
40
|
|
|
40
41
|
|
|
41
42
|
class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
|
|
@@ -47,15 +48,26 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
|
|
|
47
48
|
async def setup(self, runtime: Runtime) -> None:
|
|
48
49
|
await self.install_skills(runtime, SKILLS_DIR)
|
|
49
50
|
await ensure_node(runtime)
|
|
51
|
+
versions = {"version": self.config.version, "acp_version": ACP_VERSION}
|
|
52
|
+
directory = CLAUDE_ACP_DIR.format(**versions)
|
|
53
|
+
packages = PACKAGES_DIR.format(**versions)
|
|
54
|
+
claude_bin = CLAUDE_BIN.format(**versions)
|
|
55
|
+
acp_bin = ACP_BIN.format(**versions)
|
|
56
|
+
ready = f"{directory}/.ready"
|
|
57
|
+
script = ACP_INSTALL.replace("{packages}", packages).replace("{ready}", ready)
|
|
58
|
+
ensure = shlex.quote(
|
|
59
|
+
f"[ -f {ready} ] && [ -x {claude_bin} ] && [ -x {acp_bin} ] || ({script})"
|
|
60
|
+
)
|
|
50
61
|
acp_guarded = (
|
|
51
|
-
f"mkdir -p {
|
|
52
|
-
f'"$(command -v flock || command -v lockf)" {
|
|
53
|
-
f"sh -c {
|
|
62
|
+
f"mkdir -p {directory} && "
|
|
63
|
+
f'"$(command -v flock || command -v lockf)" {directory}/install.lock '
|
|
64
|
+
f"sh -c {ensure}"
|
|
54
65
|
)
|
|
55
66
|
acp_result = await runtime.run(
|
|
56
67
|
["sh", "-c", acp_guarded],
|
|
57
68
|
{
|
|
58
69
|
**self.config.resolved_env,
|
|
70
|
+
"VF_CLAUDE_CODE_VERSION": self.config.version,
|
|
59
71
|
"VF_CLAUDE_ACP_VERSION": ACP_VERSION,
|
|
60
72
|
},
|
|
61
73
|
)
|
|
@@ -80,6 +92,7 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
|
|
|
80
92
|
)
|
|
81
93
|
system_prompt, prompt = self.resolve_prompt(data)
|
|
82
94
|
config_dir = self.config_dir(trace)
|
|
95
|
+
versions = {"version": self.config.version, "acp_version": ACP_VERSION}
|
|
83
96
|
options: dict[str, object] = {
|
|
84
97
|
"strictMcpConfig": True,
|
|
85
98
|
"disallowedTools": self.config.disabled_tools or [],
|
|
@@ -92,6 +105,7 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
|
|
|
92
105
|
"ANTHROPIC_BASE_URL": endpoint.removesuffix("/v1"),
|
|
93
106
|
"ANTHROPIC_API_KEY": secret,
|
|
94
107
|
"ANTHROPIC_MODEL": ctx.model,
|
|
108
|
+
"CLAUDE_CODE_EXECUTABLE": CLAUDE_BIN.format(**versions),
|
|
95
109
|
"CLAUDE_CONFIG_DIR": config_dir,
|
|
96
110
|
"CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC": "1",
|
|
97
111
|
"DISABLE_AUTOUPDATER": "1",
|
|
@@ -107,7 +121,7 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
|
|
|
107
121
|
mcp_urls,
|
|
108
122
|
data,
|
|
109
123
|
env=env,
|
|
110
|
-
command=
|
|
124
|
+
command=[f"{NODE_BIN_DIR}/node", ACP_BIN.format(**versions)],
|
|
111
125
|
prompt=prompt or "",
|
|
112
126
|
session_meta=session_meta,
|
|
113
127
|
)
|
|
@@ -124,6 +138,7 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
|
|
|
124
138
|
) -> ProgramResult:
|
|
125
139
|
system_prompt, prompt = self.resolve_prompt(data)
|
|
126
140
|
config_dir = self.config_dir(trace)
|
|
141
|
+
versions = {"version": self.config.version, "acp_version": ACP_VERSION}
|
|
127
142
|
|
|
128
143
|
options: dict[str, object] = {
|
|
129
144
|
"strictMcpConfig": True,
|
|
@@ -138,6 +153,7 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
|
|
|
138
153
|
"ANTHROPIC_BASE_URL": endpoint.removesuffix("/v1"),
|
|
139
154
|
"ANTHROPIC_API_KEY": secret,
|
|
140
155
|
"ANTHROPIC_MODEL": ctx.model,
|
|
156
|
+
"CLAUDE_CODE_EXECUTABLE": CLAUDE_BIN.format(**versions),
|
|
141
157
|
"CLAUDE_CONFIG_DIR": config_dir,
|
|
142
158
|
"CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC": "1",
|
|
143
159
|
"DISABLE_AUTOUPDATER": "1",
|
|
@@ -146,7 +162,7 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
|
|
|
146
162
|
return await CLAUDE_ACP.run(
|
|
147
163
|
runtime,
|
|
148
164
|
env,
|
|
149
|
-
|
|
165
|
+
[f"{NODE_BIN_DIR}/node", ACP_BIN.format(**versions)],
|
|
150
166
|
prompt or "",
|
|
151
167
|
mcp_urls=mcp_urls,
|
|
152
168
|
session_path=f"{config_dir}/acp-session",
|
|
@@ -7,6 +7,8 @@ import re
|
|
|
7
7
|
import shlex
|
|
8
8
|
from collections import Counter
|
|
9
9
|
|
|
10
|
+
from pydantic import Field
|
|
11
|
+
|
|
10
12
|
from verifiers.v1.acp import ACP
|
|
11
13
|
from verifiers.v1.clients import ModelContext
|
|
12
14
|
from verifiers.v1.configs.harness import HarnessConfig
|
|
@@ -21,33 +23,29 @@ logger = logging.getLogger(__name__)
|
|
|
21
23
|
PROVIDER = "intercept"
|
|
22
24
|
KEY_VAR = "CODEX_INTERCEPT_KEY"
|
|
23
25
|
|
|
24
|
-
CODEX_DIR = "/var/tmp/vf-codex"
|
|
26
|
+
CODEX_DIR = "/var/tmp/vf-codex-{version}-{acp_version}"
|
|
25
27
|
PACKAGES_DIR = f"{CODEX_DIR}/acp"
|
|
26
|
-
|
|
27
|
-
|
|
28
|
+
ACP_VERSION = "1.1.10"
|
|
29
|
+
CODEX_BIN = f"{PACKAGES_DIR}/node_modules/.bin/codex"
|
|
28
30
|
ACP_BIN = f"{PACKAGES_DIR}/node_modules/.bin/codex-acp"
|
|
29
|
-
ACP_COMMAND = [f"{NODE_BIN_DIR}/node", ACP_BIN]
|
|
30
31
|
SKILLS_DIR = ".agents/skills"
|
|
31
32
|
INSTALL = r"""
|
|
32
33
|
set -e
|
|
33
34
|
export PATH="/var/tmp/vf-node/bin:$PATH"
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
&& [ -x /var/tmp/vf-codex/acp/node_modules/.bin/codex ] \
|
|
37
|
-
&& [ -x /var/tmp/vf-codex/acp/node_modules/.bin/codex-acp ]; then
|
|
38
|
-
exit 0
|
|
39
|
-
fi
|
|
40
|
-
npm install --prefix /var/tmp/vf-codex/acp --ignore-scripts --no-audit --no-fund \
|
|
35
|
+
rm -f {ready}
|
|
36
|
+
npm install --prefix {packages} --ignore-scripts --no-audit --no-fund \
|
|
41
37
|
--omit=dev \
|
|
42
38
|
"@agentclientprotocol/codex-acp@$VF_CODEX_ACP_VERSION" \
|
|
43
39
|
"@openai/codex@$VF_CODEX_VERSION" >/dev/null
|
|
44
|
-
|
|
40
|
+
touch {ready}
|
|
45
41
|
"""
|
|
46
42
|
|
|
47
43
|
CODEX_ACP = ACP()
|
|
48
44
|
|
|
49
45
|
|
|
50
46
|
class CodexHarnessConfig(HarnessConfig):
|
|
47
|
+
version: str = Field(default="0.146.1", pattern=r"^[A-Za-z0-9._+-]+$")
|
|
48
|
+
"""Codex release to install, pinned for reproducibility."""
|
|
51
49
|
multi_agent: bool = False
|
|
52
50
|
"""Enable Codex's native multi-agent v2 tools."""
|
|
53
51
|
|
|
@@ -63,19 +61,29 @@ class CodexHarness(Harness[CodexHarnessConfig]):
|
|
|
63
61
|
await ensure_node(runtime)
|
|
64
62
|
logger.info(
|
|
65
63
|
"codex: ensuring Codex %s and codex-acp %s are installed",
|
|
66
|
-
|
|
64
|
+
self.config.version,
|
|
67
65
|
ACP_VERSION,
|
|
68
66
|
)
|
|
67
|
+
versions = {"version": self.config.version, "acp_version": ACP_VERSION}
|
|
68
|
+
directory = CODEX_DIR.format(**versions)
|
|
69
|
+
packages = PACKAGES_DIR.format(**versions)
|
|
70
|
+
codex_bin = CODEX_BIN.format(**versions)
|
|
71
|
+
acp_bin = ACP_BIN.format(**versions)
|
|
72
|
+
ready = f"{directory}/.ready"
|
|
73
|
+
script = INSTALL.replace("{packages}", packages).replace("{ready}", ready)
|
|
74
|
+
ensure = shlex.quote(
|
|
75
|
+
f"[ -f {ready} ] && [ -x {codex_bin} ] && [ -x {acp_bin} ] || ({script})"
|
|
76
|
+
)
|
|
69
77
|
guarded = (
|
|
70
|
-
f"mkdir -p {
|
|
71
|
-
f'"$(command -v flock || command -v lockf)" {
|
|
72
|
-
f"sh -c {
|
|
78
|
+
f"mkdir -p {directory} && "
|
|
79
|
+
f'"$(command -v flock || command -v lockf)" {directory}/install.lock '
|
|
80
|
+
f"sh -c {ensure}"
|
|
73
81
|
)
|
|
74
82
|
install = await runtime.run(
|
|
75
83
|
["sh", "-c", guarded],
|
|
76
84
|
{
|
|
77
85
|
**self.config.resolved_env,
|
|
78
|
-
"VF_CODEX_VERSION":
|
|
86
|
+
"VF_CODEX_VERSION": self.config.version,
|
|
79
87
|
"VF_CODEX_ACP_VERSION": ACP_VERSION,
|
|
80
88
|
},
|
|
81
89
|
)
|
|
@@ -112,7 +120,10 @@ class CodexHarness(Harness[CodexHarnessConfig]):
|
|
|
112
120
|
{},
|
|
113
121
|
data,
|
|
114
122
|
env=env,
|
|
115
|
-
command=
|
|
123
|
+
command=[
|
|
124
|
+
f"{NODE_BIN_DIR}/node",
|
|
125
|
+
ACP_BIN.format(version=self.config.version, acp_version=ACP_VERSION),
|
|
126
|
+
],
|
|
116
127
|
prompt=prompt,
|
|
117
128
|
system_prompt=system_prompt,
|
|
118
129
|
)
|
|
@@ -139,7 +150,10 @@ class CodexHarness(Harness[CodexHarnessConfig]):
|
|
|
139
150
|
return await CODEX_ACP.run(
|
|
140
151
|
runtime,
|
|
141
152
|
env,
|
|
142
|
-
|
|
153
|
+
[
|
|
154
|
+
f"{NODE_BIN_DIR}/node",
|
|
155
|
+
ACP_BIN.format(version=self.config.version, acp_version=ACP_VERSION),
|
|
156
|
+
],
|
|
143
157
|
prompt,
|
|
144
158
|
system_prompt=system_prompt,
|
|
145
159
|
session_path=f"{self.trace_home(trace)}/acp-session",
|
|
@@ -4,6 +4,8 @@ import json
|
|
|
4
4
|
import logging
|
|
5
5
|
import shlex
|
|
6
6
|
|
|
7
|
+
from pydantic import Field
|
|
8
|
+
|
|
7
9
|
from verifiers.v1.acp import ACP
|
|
8
10
|
from verifiers.v1.clients import ModelContext
|
|
9
11
|
from verifiers.v1.configs.harness import HarnessConfig
|
|
@@ -43,7 +45,7 @@ KIMI_ACP = ACP()
|
|
|
43
45
|
|
|
44
46
|
|
|
45
47
|
class KimiCodeHarnessConfig(HarnessConfig):
|
|
46
|
-
version: str = "0.
|
|
48
|
+
version: str = Field(default="0.34.0", pattern=r"^[A-Za-z0-9._+-]+$")
|
|
47
49
|
"""Kimi Code release to install, pinned for reproducibility."""
|
|
48
50
|
|
|
49
51
|
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
from pathlib import Path
|
|
2
2
|
|
|
3
|
+
from pydantic import Field
|
|
4
|
+
|
|
3
5
|
from verifiers.v1.clients import ModelContext
|
|
4
6
|
from verifiers.v1.configs.harness import HarnessConfig
|
|
5
7
|
from verifiers.v1.harness import Harness
|
|
@@ -11,7 +13,7 @@ PROGRAM_SOURCE = (Path(__file__).resolve().parent / "program.py").read_text()
|
|
|
11
13
|
|
|
12
14
|
|
|
13
15
|
class MiniSWEAgentHarnessConfig(HarnessConfig):
|
|
14
|
-
version: str = "2.4.
|
|
16
|
+
version: str = Field(default="2.4.6", pattern=r"^[A-Za-z0-9._+-]+$")
|
|
15
17
|
"""mini-swe-agent release to install, pinned for reproducibility."""
|
|
16
18
|
|
|
17
19
|
|
|
@@ -117,10 +117,8 @@ class OpenClawHarness(Harness[OpenClawHarnessConfig]):
|
|
|
117
117
|
data: TaskData,
|
|
118
118
|
) -> ProgramResult:
|
|
119
119
|
system_prompt, prompt = self.resolve_prompt(data)
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
"none",
|
|
123
|
-
) or ctx.model.rsplit("/", 1)[-1].startswith(("gpt-5", "o1", "o3", "o4"))
|
|
120
|
+
# Opt-in via sampling: OpenClaw redacts persisted encrypted reasoning, so resumed sessions 400 replaying it.
|
|
121
|
+
reasoning = ctx.sampling.reasoning_effort not in (None, "none")
|
|
124
122
|
directory = OPENCLAW_DIR.format(version=self.config.version)
|
|
125
123
|
state_dir = f".vf-openclaw/{trace.id}"
|
|
126
124
|
config_path = f"{state_dir}/openclaw.json"
|
|
@@ -4,6 +4,8 @@ import json
|
|
|
4
4
|
import logging
|
|
5
5
|
import shlex
|
|
6
6
|
|
|
7
|
+
from pydantic import Field
|
|
8
|
+
|
|
7
9
|
from verifiers.v1.acp import ACP
|
|
8
10
|
from verifiers.v1.clients import ModelContext
|
|
9
11
|
from verifiers.v1.configs.harness import HarnessConfig
|
|
@@ -23,8 +25,8 @@ PI_DIR = "/var/tmp/vf-pi"
|
|
|
23
25
|
PACKAGES_DIR = f"{PI_DIR}/mcp"
|
|
24
26
|
PI_BIN = f"{PACKAGES_DIR}/node_modules/.bin/pi"
|
|
25
27
|
SKILLS_DIR = ".agents/skills"
|
|
26
|
-
MCP_VERSION = "2.
|
|
27
|
-
ACP_VERSION = "0.0.
|
|
28
|
+
MCP_VERSION = "2.20.1"
|
|
29
|
+
ACP_VERSION = "0.0.33"
|
|
28
30
|
MCP_ADAPTER = f"{PACKAGES_DIR}/node_modules/pi-mcp-adapter/index.ts"
|
|
29
31
|
ACP_BIN = f"{PACKAGES_DIR}/node_modules/.bin/pi-acp"
|
|
30
32
|
ACP_COMMAND = [
|
|
@@ -86,7 +88,7 @@ PI_ACP = ACP()
|
|
|
86
88
|
|
|
87
89
|
|
|
88
90
|
class PiHarnessConfig(HarnessConfig):
|
|
89
|
-
version: str = "0.
|
|
91
|
+
version: str = Field(default="0.84.0", pattern=r"^[A-Za-z0-9._+-]+$")
|
|
90
92
|
"""Pi release to install, pinned for reproducibility."""
|
|
91
93
|
|
|
92
94
|
|
|
@@ -223,4 +225,6 @@ class PiHarness(Harness[PiHarnessConfig]):
|
|
|
223
225
|
ACP_COMMAND,
|
|
224
226
|
prompt,
|
|
225
227
|
session_path=f"{agent_dir}/acp-session",
|
|
228
|
+
# Pi can end after its final tool completes without a text message.
|
|
229
|
+
allow_empty_tool_reply=True,
|
|
226
230
|
)
|
|
@@ -31,7 +31,7 @@ POOL_ACP = ACP()
|
|
|
31
31
|
|
|
32
32
|
|
|
33
33
|
class PoolHarnessConfig(HarnessConfig):
|
|
34
|
-
version: str = Field(default="1.0.
|
|
34
|
+
version: str = Field(default="1.0.15", pattern=r"^[A-Za-z0-9._+-]+$")
|
|
35
35
|
"""Pool release to install, pinned for reproducibility."""
|
|
36
36
|
|
|
37
37
|
|
|
@@ -30,7 +30,7 @@ RLM_ACP = ACP()
|
|
|
30
30
|
|
|
31
31
|
|
|
32
32
|
class RLMHarnessConfig(HarnessConfig):
|
|
33
|
-
version: str = "main"
|
|
33
|
+
version: str = Field(default="main", min_length=1)
|
|
34
34
|
"""Git ref (branch, tag, or commit) of rlm-harness to install."""
|
|
35
35
|
max_depth: int = 0
|
|
36
36
|
"""Recursion depth rlm may spawn sub-harnesses to (RLM_MAX_DEPTH)."""
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
import logging
|
|
2
2
|
from pathlib import Path
|
|
3
3
|
|
|
4
|
+
from pydantic import Field
|
|
5
|
+
|
|
4
6
|
from verifiers.v1.clients import ModelContext
|
|
5
7
|
from verifiers.v1.configs.harness import HarnessConfig
|
|
6
8
|
from verifiers.v1.harness import Harness
|
|
@@ -13,7 +15,7 @@ logger = logging.getLogger(__name__)
|
|
|
13
15
|
|
|
14
16
|
|
|
15
17
|
class Terminus2HarnessConfig(HarnessConfig):
|
|
16
|
-
version: str = "0.20.0"
|
|
18
|
+
version: str = Field(default="0.20.0", pattern=r"^[A-Za-z0-9._+-]+$")
|
|
17
19
|
"""Harbor release to install, pinned for reproducibility."""
|
|
18
20
|
|
|
19
21
|
|
|
@@ -9,7 +9,9 @@ SSE requests are supported.
|
|
|
9
9
|
One server multiplexes many rollouts: each rollout registers separate model and state
|
|
10
10
|
capabilities, and the server routes each to the right session. So N rollouts need one
|
|
11
11
|
server (and, behind a remote runtime, one tunnel) per pool member rather than one each —
|
|
12
|
-
see `interception.pool`.
|
|
12
|
+
see `interception.pool`. The server also owns the model clients (one per distinct endpoint
|
|
13
|
+
config, assigned to each session at register and closed with the server), so its rollouts
|
|
14
|
+
share one bounded keepalive connection pool upstream instead of churning per-rollout TCP.
|
|
13
15
|
|
|
14
16
|
The server is a pure model boundary: one request, one turn — refusal checks (limits,
|
|
15
17
|
`@stop`s), the model call, the graph commit, retry atomicity. A run's user exchange
|
|
@@ -34,6 +36,8 @@ from pydantic import ValidationError
|
|
|
34
36
|
from pydantic_core import PydanticSerializationError, from_json, to_json
|
|
35
37
|
|
|
36
38
|
from verifiers.v1 import graph
|
|
39
|
+
from verifiers.v1.clients import Client, resolve_client
|
|
40
|
+
from verifiers.v1.configs.client import BaseClientConfig
|
|
37
41
|
from verifiers.v1.dialects import DIALECTS, Dialect
|
|
38
42
|
from verifiers.v1.dialects.base import is_sse_done_event
|
|
39
43
|
from verifiers.v1.errors import (
|
|
@@ -130,6 +134,7 @@ class InterceptionServer(Interception):
|
|
|
130
134
|
) -> None:
|
|
131
135
|
super().__init__()
|
|
132
136
|
self.sessions: dict[str, RolloutSession] = {}
|
|
137
|
+
self.clients: dict[str, Client] = {}
|
|
133
138
|
self.state_sessions: dict[str, RolloutSession] = {}
|
|
134
139
|
self.state_routes: dict[str, RolloutSession] = {}
|
|
135
140
|
self.state_service_secrets = frozenset(state_service_secrets)
|
|
@@ -147,8 +152,22 @@ class InterceptionServer(Interception):
|
|
|
147
152
|
"""Rollouts currently registered — what the pools balance on."""
|
|
148
153
|
return len(self.sessions)
|
|
149
154
|
|
|
155
|
+
def _client(self, config: BaseClientConfig) -> Client:
|
|
156
|
+
"""The server-owned client for `config` — one per distinct endpoint config, shared
|
|
157
|
+
by every session registered under it, so the rollouts this server multiplexes reuse
|
|
158
|
+
one bounded keepalive pool instead of each opening (and tearing down) their own
|
|
159
|
+
connections. Closed with the server."""
|
|
160
|
+
key = config.model_dump_json()
|
|
161
|
+
client = self.clients.get(key)
|
|
162
|
+
if client is None:
|
|
163
|
+
client = self.clients[key] = resolve_client(config)
|
|
164
|
+
self.stack.push_async_callback(client.close)
|
|
165
|
+
return client
|
|
166
|
+
|
|
150
167
|
def register(self, session: RolloutSession) -> tuple[str, str]:
|
|
151
|
-
"""Register separate capabilities for model inference and private task state
|
|
168
|
+
"""Register separate capabilities for model inference and private task state, and
|
|
169
|
+
assign the session its server-owned model client."""
|
|
170
|
+
session.client = self._client(session.ctx.client)
|
|
152
171
|
model_secret = secrets.token_urlsafe(16)
|
|
153
172
|
state_secret = secrets.token_urlsafe(16)
|
|
154
173
|
self.sessions[model_secret] = session
|