verifiers 0.2.2.dev96__py3-none-any.whl → 0.2.2.dev112__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. verifiers/v1/acp/__init__.py +6 -2
  2. verifiers/v1/clients/train.py +6 -2
  3. verifiers/v1/configs/client.py +3 -2
  4. verifiers/v1/dialects/chat.py +5 -1
  5. verifiers/v1/envs/agentic_judge/env.py +5 -3
  6. verifiers/v1/envs/user_sim/env.py +2 -1
  7. verifiers/v1/gepa/runner.py +1 -1
  8. verifiers/v1/harnesses/claude_code/harness.py +31 -15
  9. verifiers/v1/harnesses/codex/harness.py +33 -19
  10. verifiers/v1/harnesses/kimi_code/harness.py +3 -1
  11. verifiers/v1/harnesses/mini_swe_agent/harness.py +3 -1
  12. verifiers/v1/harnesses/openclaw/harness.py +2 -4
  13. verifiers/v1/harnesses/pi/harness.py +7 -3
  14. verifiers/v1/harnesses/pool/harness.py +1 -1
  15. verifiers/v1/harnesses/rlm/harness.py +1 -1
  16. verifiers/v1/harnesses/terminus_2/harness.py +3 -1
  17. verifiers/v1/interception/server.py +21 -2
  18. verifiers/v1/mcp/launch.py +108 -41
  19. verifiers/v1/mcp/server.py +3 -2
  20. verifiers/v1/mcp/toolset.py +1 -1
  21. verifiers/v1/rollout.py +4 -13
  22. verifiers/v1/runtimes/base.py +9 -0
  23. verifiers/v1/runtimes/limiters.py +9 -4
  24. verifiers/v1/session.py +6 -3
  25. verifiers/v1/tasksets/__init__.py +3 -0
  26. verifiers/v1/tasksets/nemo_gym/__init__.py +17 -0
  27. verifiers/v1/tasksets/nemo_gym/response.py +110 -0
  28. verifiers/v1/tasksets/nemo_gym/server.py +42 -0
  29. verifiers/v1/tasksets/nemo_gym/taskset.py +170 -0
  30. verifiers/v1/tasksets/nemo_gym/toolset.py +107 -0
  31. {verifiers-0.2.2.dev96.dist-info → verifiers-0.2.2.dev112.dist-info}/METADATA +4 -2
  32. {verifiers-0.2.2.dev96.dist-info → verifiers-0.2.2.dev112.dist-info}/RECORD +35 -30
  33. {verifiers-0.2.2.dev96.dist-info → verifiers-0.2.2.dev112.dist-info}/WHEEL +0 -0
  34. {verifiers-0.2.2.dev96.dist-info → verifiers-0.2.2.dev112.dist-info}/entry_points.txt +0 -0
  35. {verifiers-0.2.2.dev96.dist-info → verifiers-0.2.2.dev112.dist-info}/licenses/LICENSE +0 -0
@@ -121,7 +121,9 @@ class ACP:
121
121
  "allow_empty_tool_reply": allow_empty_tool_reply,
122
122
  }
123
123
  program = await runtime.prepare_uv_script(
124
- ACP_SOURCE, {**env, "UV_FROZEN": "false"}
124
+ ACP_SOURCE,
125
+ {**env, "UV_FROZEN": "false"},
126
+ activate=False,
125
127
  )
126
128
  directory = f".vf-acp-{secrets.token_hex(8)}"
127
129
  created = await runtime.run(["mkdir", "-m", "700", directory], {})
@@ -198,7 +200,9 @@ class ACPHarnessSession(HarnessSession):
198
200
  async def _start(self) -> None:
199
201
  self._stderr_tail.clear()
200
202
  program = await self.runtime.prepare_uv_script(
201
- ACP_SOURCE, {**self.env, "UV_FROZEN": "false"}
203
+ ACP_SOURCE,
204
+ {**self.env, "UV_FROZEN": "false"},
205
+ activate=False,
202
206
  )
203
207
  process = await self.runtime.open_process([*program, "stream"], self.env)
204
208
  self._process = process
@@ -12,6 +12,7 @@ from typing import Any, ClassVar, TypeVar
12
12
  from openai import OpenAIError
13
13
  from renderers import OverlongPromptError as RendererOverlongPromptError
14
14
  from renderers import RenderedTokens, Renderer, RendererConfig
15
+ from renderers.base import ToolCallParseStatus
15
16
 
16
17
  from verifiers.v1.clients.base import build_async_openai
17
18
  from verifiers.v1.clients.client import SESSION_ID_HEADER, Client
@@ -115,6 +116,8 @@ def response_from_generate(
115
116
  )
116
117
  for i, tc in enumerate(result.get("tool_calls") or [])
117
118
  if getattr(tc, "name", None)
119
+ # TODO: we need a better way for renderers to expose this
120
+ and getattr(tc, "status", None) != ToolCallParseStatus.UNKNOWN_TOOL
118
121
  ] or None
119
122
  prompt_ids = result.get("prompt_ids") or []
120
123
  completion_ids = result.get("completion_ids") or []
@@ -296,8 +299,9 @@ class ElasticRendererPool:
296
299
  class TrainClient(Client):
297
300
  """Renders prompts to token ids and calls a vLLM `/inference/v1/generate` engine.
298
301
 
299
- One client per rollout: it owns its engine connection and takes a slot on the shared
300
- `ElasticRendererPool` for each turn."""
302
+ Owned by the interception server and shared by the rollouts it multiplexes: they reuse
303
+ its engine connection pool, and each turn takes a slot on the shared
304
+ `ElasticRendererPool`."""
301
305
 
302
306
  def __init__(self, config: TrainClientConfig) -> None:
303
307
  self.config = config
@@ -1,8 +1,9 @@
1
1
  """Client configs: describe an OpenAI-compatible endpoint.
2
2
 
3
3
  A `BaseClientConfig` is an OpenAI-compatible endpoint (base_url + API-key env var
4
- + extra headers); `clients.resolve_client` turns one into a live `Client`, and every
5
- rollout does so for itself. The default Prime endpoint, API key, and team fall back to
4
+ + extra headers); `clients.resolve_client` turns one into a live `Client` — the
5
+ interception server builds one per distinct config and shares it across the rollouts
6
+ it multiplexes. The default Prime endpoint, API key, and team fall back to
6
7
  the active Prime CLI config, so direct `uv run eval` calls behave like `prime eval`.
7
8
  Both the eval entrypoint (its model client) and in-env LLM calls (e.g. a judge reward)
8
9
  build clients from these. `ClientConfig` is the CLI-selectable discriminated union
@@ -141,7 +141,11 @@ def _content_to_wire(content):
141
141
 
142
142
  def message_to_wire(message: Message) -> dict:
143
143
  if message.role == "assistant":
144
- wire: dict = {"role": "assistant", "content": message.content}
144
+ # Strict providers reject `content: null` without tool calls.
145
+ content = message.content
146
+ if content is None and not message.tool_calls:
147
+ content = ""
148
+ wire: dict = {"role": "assistant", "content": content}
145
149
  if message.provider_state:
146
150
  wire["reasoning_details"] = message.provider_state
147
151
  elif message.reasoning_content is not None:
@@ -309,8 +309,6 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
309
309
 
310
310
  async def finalize(self, task: vf.Task, episode: vf.Episode) -> None:
311
311
  by_agent = {t.agent.name: t for t in episode.traces}
312
- if "judge" not in by_agent:
313
- return
314
312
  solution, verdict = by_agent["solver"], by_agent["judge"]
315
313
  verdicts = RubricVerdicts.model_validate(verdict.info.get("verdict")).verdicts
316
314
  criteria = self.config.task.criteria()
@@ -332,6 +330,10 @@ class SharedAgenticJudgeEnv(AgenticJudgeEnv):
332
330
  async def run(self, task: vf.Task, agents: vf.Agents) -> None:
333
331
  async with agents.solver.provision(task) as box:
334
332
  solution = await agents.solver.run(task, runtime=box)
333
+ if not solution.ok:
334
+ raise RuntimeError(
335
+ "the solver's rollout failed, so the judge never ran"
336
+ )
335
337
  judge_task = JudgeTask.from_trace(solution, self.config.task)
336
338
  await agents.judge.run(judge_task, runtime=box)
337
339
 
@@ -342,7 +344,7 @@ class IsolatedAgenticJudgeEnv(AgenticJudgeEnv):
342
344
  async def run(self, task: vf.Task, agents: vf.Agents) -> None:
343
345
  solution = await agents.solver.run(task)
344
346
  if not solution.ok:
345
- return
347
+ raise RuntimeError("the solver's rollout failed, so the judge never ran")
346
348
  await agents.judge.run(
347
349
  JudgeTask.from_trace(solution, self.config.task, share_runtime=False)
348
350
  )
@@ -21,13 +21,14 @@ from pydantic import Field
21
21
 
22
22
  import verifiers.v1 as vf
23
23
 
24
- PERSONA = """You are role-playing a USER talking to an AI assistant. This is your situation and goal:
24
+ PERSONA = """You are role-playing a USER talking to an AI assistant. This is your situation — what you want the ASSISTANT to do for you:
25
25
 
26
26
  {scenario}
27
27
 
28
28
  Rules:
29
29
  - Open the conversation with your request, in your own words.
30
30
  - Stay in character: short, natural user messages; never act as the assistant.
31
+ - The assistant does the work, not you: never perform the task yourself — any tool call, answer, or output format it asks for must come from the assistant; ask for it instead.
31
32
  - Reveal details only when asked, as a real user would.
32
33
  - When the assistant has fully met your goal — or you are convinced it cannot — reply with exactly {done} and nothing else."""
33
34
 
@@ -73,7 +73,7 @@ def run_gepa(env: Env, config: GEPAConfig) -> GEPAResult:
73
73
  await append_episode(run_dir, episode, write_lock)
74
74
 
75
75
  try:
76
- # The endpoint stays config: each rollout builds and closes its own client.
76
+ # The endpoint stays config: the interception server builds the live client.
77
77
  ctx = ModelContext(
78
78
  client=config.client, model=config.model, sampling=config.sampling
79
79
  )
@@ -2,6 +2,8 @@
2
2
 
3
3
  import shlex
4
4
 
5
+ from pydantic import Field
6
+
5
7
  from verifiers.v1.acp import ACP
6
8
  from verifiers.v1.clients import ModelContext
7
9
  from verifiers.v1.configs.harness import HarnessConfig
@@ -11,31 +13,30 @@ from verifiers.v1.runtimes import ProgramResult, Runtime
11
13
  from verifiers.v1.task import TaskData
12
14
  from verifiers.v1.trace import Trace
13
15
 
14
- CLAUDE_ACP_DIR = "/var/tmp/vf-claude-agent-acp"
16
+ CLAUDE_ACP_DIR = "/var/tmp/vf-claude-agent-acp-{version}-{acp_version}"
15
17
  PACKAGES_DIR = f"{CLAUDE_ACP_DIR}/packages"
16
- ACP_VERSION = "0.63.0"
18
+ ACP_VERSION = "0.65.0"
19
+ CLAUDE_BIN = f"{PACKAGES_DIR}/node_modules/.bin/claude"
17
20
  ACP_BIN = f"{PACKAGES_DIR}/node_modules/.bin/claude-agent-acp"
18
- ACP_COMMAND = [f"{NODE_BIN_DIR}/node", ACP_BIN]
19
21
  CLAUDE_CONFIG_ROOT = ".vf-claude"
20
22
  SKILLS_DIR = ".claude/skills"
21
23
  ACP_INSTALL = r"""
22
24
  set -e
23
25
  export PATH="/var/tmp/vf-node/bin:$PATH"
24
- if [ "$(cat /var/tmp/vf-claude-agent-acp/.version 2>/dev/null)" = "$VF_CLAUDE_ACP_VERSION" ] \
25
- && [ -x /var/tmp/vf-claude-agent-acp/packages/node_modules/.bin/claude-agent-acp ]; then
26
- exit 0
27
- fi
28
- npm install --prefix /var/tmp/vf-claude-agent-acp/packages --ignore-scripts --no-audit --no-fund \
26
+ rm -f {ready}
27
+ npm install --prefix {packages} --no-audit --no-fund \
29
28
  --omit=dev \
29
+ "@anthropic-ai/claude-code@$VF_CLAUDE_CODE_VERSION" \
30
30
  "@agentclientprotocol/claude-agent-acp@$VF_CLAUDE_ACP_VERSION" >/dev/null
31
- printf %s "$VF_CLAUDE_ACP_VERSION" > /var/tmp/vf-claude-agent-acp/.version
31
+ touch {ready}
32
32
  """
33
33
 
34
34
  CLAUDE_ACP = ACP()
35
35
 
36
36
 
37
37
  class ClaudeCodeHarnessConfig(HarnessConfig):
38
- pass
38
+ version: str = Field(default="2.1.223", pattern=r"^[A-Za-z0-9._+-]+$")
39
+ """Claude Code release to install, pinned for reproducibility."""
39
40
 
40
41
 
41
42
  class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
@@ -47,15 +48,26 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
47
48
  async def setup(self, runtime: Runtime) -> None:
48
49
  await self.install_skills(runtime, SKILLS_DIR)
49
50
  await ensure_node(runtime)
51
+ versions = {"version": self.config.version, "acp_version": ACP_VERSION}
52
+ directory = CLAUDE_ACP_DIR.format(**versions)
53
+ packages = PACKAGES_DIR.format(**versions)
54
+ claude_bin = CLAUDE_BIN.format(**versions)
55
+ acp_bin = ACP_BIN.format(**versions)
56
+ ready = f"{directory}/.ready"
57
+ script = ACP_INSTALL.replace("{packages}", packages).replace("{ready}", ready)
58
+ ensure = shlex.quote(
59
+ f"[ -f {ready} ] && [ -x {claude_bin} ] && [ -x {acp_bin} ] || ({script})"
60
+ )
50
61
  acp_guarded = (
51
- f"mkdir -p {CLAUDE_ACP_DIR} && "
52
- f'"$(command -v flock || command -v lockf)" {CLAUDE_ACP_DIR}/install.lock '
53
- f"sh -c {shlex.quote(ACP_INSTALL)}"
62
+ f"mkdir -p {directory} && "
63
+ f'"$(command -v flock || command -v lockf)" {directory}/install.lock '
64
+ f"sh -c {ensure}"
54
65
  )
55
66
  acp_result = await runtime.run(
56
67
  ["sh", "-c", acp_guarded],
57
68
  {
58
69
  **self.config.resolved_env,
70
+ "VF_CLAUDE_CODE_VERSION": self.config.version,
59
71
  "VF_CLAUDE_ACP_VERSION": ACP_VERSION,
60
72
  },
61
73
  )
@@ -80,6 +92,7 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
80
92
  )
81
93
  system_prompt, prompt = self.resolve_prompt(data)
82
94
  config_dir = self.config_dir(trace)
95
+ versions = {"version": self.config.version, "acp_version": ACP_VERSION}
83
96
  options: dict[str, object] = {
84
97
  "strictMcpConfig": True,
85
98
  "disallowedTools": self.config.disabled_tools or [],
@@ -92,6 +105,7 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
92
105
  "ANTHROPIC_BASE_URL": endpoint.removesuffix("/v1"),
93
106
  "ANTHROPIC_API_KEY": secret,
94
107
  "ANTHROPIC_MODEL": ctx.model,
108
+ "CLAUDE_CODE_EXECUTABLE": CLAUDE_BIN.format(**versions),
95
109
  "CLAUDE_CONFIG_DIR": config_dir,
96
110
  "CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC": "1",
97
111
  "DISABLE_AUTOUPDATER": "1",
@@ -107,7 +121,7 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
107
121
  mcp_urls,
108
122
  data,
109
123
  env=env,
110
- command=ACP_COMMAND,
124
+ command=[f"{NODE_BIN_DIR}/node", ACP_BIN.format(**versions)],
111
125
  prompt=prompt or "",
112
126
  session_meta=session_meta,
113
127
  )
@@ -124,6 +138,7 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
124
138
  ) -> ProgramResult:
125
139
  system_prompt, prompt = self.resolve_prompt(data)
126
140
  config_dir = self.config_dir(trace)
141
+ versions = {"version": self.config.version, "acp_version": ACP_VERSION}
127
142
 
128
143
  options: dict[str, object] = {
129
144
  "strictMcpConfig": True,
@@ -138,6 +153,7 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
138
153
  "ANTHROPIC_BASE_URL": endpoint.removesuffix("/v1"),
139
154
  "ANTHROPIC_API_KEY": secret,
140
155
  "ANTHROPIC_MODEL": ctx.model,
156
+ "CLAUDE_CODE_EXECUTABLE": CLAUDE_BIN.format(**versions),
141
157
  "CLAUDE_CONFIG_DIR": config_dir,
142
158
  "CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC": "1",
143
159
  "DISABLE_AUTOUPDATER": "1",
@@ -146,7 +162,7 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
146
162
  return await CLAUDE_ACP.run(
147
163
  runtime,
148
164
  env,
149
- ACP_COMMAND,
165
+ [f"{NODE_BIN_DIR}/node", ACP_BIN.format(**versions)],
150
166
  prompt or "",
151
167
  mcp_urls=mcp_urls,
152
168
  session_path=f"{config_dir}/acp-session",
@@ -7,6 +7,8 @@ import re
7
7
  import shlex
8
8
  from collections import Counter
9
9
 
10
+ from pydantic import Field
11
+
10
12
  from verifiers.v1.acp import ACP
11
13
  from verifiers.v1.clients import ModelContext
12
14
  from verifiers.v1.configs.harness import HarnessConfig
@@ -21,33 +23,29 @@ logger = logging.getLogger(__name__)
21
23
  PROVIDER = "intercept"
22
24
  KEY_VAR = "CODEX_INTERCEPT_KEY"
23
25
 
24
- CODEX_DIR = "/var/tmp/vf-codex"
26
+ CODEX_DIR = "/var/tmp/vf-codex-{version}-{acp_version}"
25
27
  PACKAGES_DIR = f"{CODEX_DIR}/acp"
26
- CODEX_VERSION = "0.145.0"
27
- ACP_VERSION = "1.1.7"
28
+ ACP_VERSION = "1.1.10"
29
+ CODEX_BIN = f"{PACKAGES_DIR}/node_modules/.bin/codex"
28
30
  ACP_BIN = f"{PACKAGES_DIR}/node_modules/.bin/codex-acp"
29
- ACP_COMMAND = [f"{NODE_BIN_DIR}/node", ACP_BIN]
30
31
  SKILLS_DIR = ".agents/skills"
31
32
  INSTALL = r"""
32
33
  set -e
33
34
  export PATH="/var/tmp/vf-node/bin:$PATH"
34
- versions="$VF_CODEX_VERSION:$VF_CODEX_ACP_VERSION"
35
- if [ "$(cat /var/tmp/vf-codex/.versions 2>/dev/null)" = "$versions" ] \
36
- && [ -x /var/tmp/vf-codex/acp/node_modules/.bin/codex ] \
37
- && [ -x /var/tmp/vf-codex/acp/node_modules/.bin/codex-acp ]; then
38
- exit 0
39
- fi
40
- npm install --prefix /var/tmp/vf-codex/acp --ignore-scripts --no-audit --no-fund \
35
+ rm -f {ready}
36
+ npm install --prefix {packages} --ignore-scripts --no-audit --no-fund \
41
37
  --omit=dev \
42
38
  "@agentclientprotocol/codex-acp@$VF_CODEX_ACP_VERSION" \
43
39
  "@openai/codex@$VF_CODEX_VERSION" >/dev/null
44
- printf %s "$versions" > /var/tmp/vf-codex/.versions
40
+ touch {ready}
45
41
  """
46
42
 
47
43
  CODEX_ACP = ACP()
48
44
 
49
45
 
50
46
  class CodexHarnessConfig(HarnessConfig):
47
+ version: str = Field(default="0.146.1", pattern=r"^[A-Za-z0-9._+-]+$")
48
+ """Codex release to install, pinned for reproducibility."""
51
49
  multi_agent: bool = False
52
50
  """Enable Codex's native multi-agent v2 tools."""
53
51
 
@@ -63,19 +61,29 @@ class CodexHarness(Harness[CodexHarnessConfig]):
63
61
  await ensure_node(runtime)
64
62
  logger.info(
65
63
  "codex: ensuring Codex %s and codex-acp %s are installed",
66
- CODEX_VERSION,
64
+ self.config.version,
67
65
  ACP_VERSION,
68
66
  )
67
+ versions = {"version": self.config.version, "acp_version": ACP_VERSION}
68
+ directory = CODEX_DIR.format(**versions)
69
+ packages = PACKAGES_DIR.format(**versions)
70
+ codex_bin = CODEX_BIN.format(**versions)
71
+ acp_bin = ACP_BIN.format(**versions)
72
+ ready = f"{directory}/.ready"
73
+ script = INSTALL.replace("{packages}", packages).replace("{ready}", ready)
74
+ ensure = shlex.quote(
75
+ f"[ -f {ready} ] && [ -x {codex_bin} ] && [ -x {acp_bin} ] || ({script})"
76
+ )
69
77
  guarded = (
70
- f"mkdir -p {CODEX_DIR} && "
71
- f'"$(command -v flock || command -v lockf)" {CODEX_DIR}/install.lock '
72
- f"sh -c {shlex.quote(INSTALL)}"
78
+ f"mkdir -p {directory} && "
79
+ f'"$(command -v flock || command -v lockf)" {directory}/install.lock '
80
+ f"sh -c {ensure}"
73
81
  )
74
82
  install = await runtime.run(
75
83
  ["sh", "-c", guarded],
76
84
  {
77
85
  **self.config.resolved_env,
78
- "VF_CODEX_VERSION": CODEX_VERSION,
86
+ "VF_CODEX_VERSION": self.config.version,
79
87
  "VF_CODEX_ACP_VERSION": ACP_VERSION,
80
88
  },
81
89
  )
@@ -112,7 +120,10 @@ class CodexHarness(Harness[CodexHarnessConfig]):
112
120
  {},
113
121
  data,
114
122
  env=env,
115
- command=ACP_COMMAND,
123
+ command=[
124
+ f"{NODE_BIN_DIR}/node",
125
+ ACP_BIN.format(version=self.config.version, acp_version=ACP_VERSION),
126
+ ],
116
127
  prompt=prompt,
117
128
  system_prompt=system_prompt,
118
129
  )
@@ -139,7 +150,10 @@ class CodexHarness(Harness[CodexHarnessConfig]):
139
150
  return await CODEX_ACP.run(
140
151
  runtime,
141
152
  env,
142
- ACP_COMMAND,
153
+ [
154
+ f"{NODE_BIN_DIR}/node",
155
+ ACP_BIN.format(version=self.config.version, acp_version=ACP_VERSION),
156
+ ],
143
157
  prompt,
144
158
  system_prompt=system_prompt,
145
159
  session_path=f"{self.trace_home(trace)}/acp-session",
@@ -4,6 +4,8 @@ import json
4
4
  import logging
5
5
  import shlex
6
6
 
7
+ from pydantic import Field
8
+
7
9
  from verifiers.v1.acp import ACP
8
10
  from verifiers.v1.clients import ModelContext
9
11
  from verifiers.v1.configs.harness import HarnessConfig
@@ -43,7 +45,7 @@ KIMI_ACP = ACP()
43
45
 
44
46
 
45
47
  class KimiCodeHarnessConfig(HarnessConfig):
46
- version: str = "0.29.0"
48
+ version: str = Field(default="0.34.0", pattern=r"^[A-Za-z0-9._+-]+$")
47
49
  """Kimi Code release to install, pinned for reproducibility."""
48
50
 
49
51
 
@@ -1,5 +1,7 @@
1
1
  from pathlib import Path
2
2
 
3
+ from pydantic import Field
4
+
3
5
  from verifiers.v1.clients import ModelContext
4
6
  from verifiers.v1.configs.harness import HarnessConfig
5
7
  from verifiers.v1.harness import Harness
@@ -11,7 +13,7 @@ PROGRAM_SOURCE = (Path(__file__).resolve().parent / "program.py").read_text()
11
13
 
12
14
 
13
15
  class MiniSWEAgentHarnessConfig(HarnessConfig):
14
- version: str = "2.4.5"
16
+ version: str = Field(default="2.4.6", pattern=r"^[A-Za-z0-9._+-]+$")
15
17
  """mini-swe-agent release to install, pinned for reproducibility."""
16
18
 
17
19
 
@@ -117,10 +117,8 @@ class OpenClawHarness(Harness[OpenClawHarnessConfig]):
117
117
  data: TaskData,
118
118
  ) -> ProgramResult:
119
119
  system_prompt, prompt = self.resolve_prompt(data)
120
- reasoning = ctx.sampling.reasoning_effort not in (
121
- None,
122
- "none",
123
- ) or ctx.model.rsplit("/", 1)[-1].startswith(("gpt-5", "o1", "o3", "o4"))
120
+ # Opt-in via sampling: OpenClaw redacts persisted encrypted reasoning, so resumed sessions 400 replaying it.
121
+ reasoning = ctx.sampling.reasoning_effort not in (None, "none")
124
122
  directory = OPENCLAW_DIR.format(version=self.config.version)
125
123
  state_dir = f".vf-openclaw/{trace.id}"
126
124
  config_path = f"{state_dir}/openclaw.json"
@@ -4,6 +4,8 @@ import json
4
4
  import logging
5
5
  import shlex
6
6
 
7
+ from pydantic import Field
8
+
7
9
  from verifiers.v1.acp import ACP
8
10
  from verifiers.v1.clients import ModelContext
9
11
  from verifiers.v1.configs.harness import HarnessConfig
@@ -23,8 +25,8 @@ PI_DIR = "/var/tmp/vf-pi"
23
25
  PACKAGES_DIR = f"{PI_DIR}/mcp"
24
26
  PI_BIN = f"{PACKAGES_DIR}/node_modules/.bin/pi"
25
27
  SKILLS_DIR = ".agents/skills"
26
- MCP_VERSION = "2.11.0"
27
- ACP_VERSION = "0.0.31"
28
+ MCP_VERSION = "2.20.1"
29
+ ACP_VERSION = "0.0.33"
28
30
  MCP_ADAPTER = f"{PACKAGES_DIR}/node_modules/pi-mcp-adapter/index.ts"
29
31
  ACP_BIN = f"{PACKAGES_DIR}/node_modules/.bin/pi-acp"
30
32
  ACP_COMMAND = [
@@ -86,7 +88,7 @@ PI_ACP = ACP()
86
88
 
87
89
 
88
90
  class PiHarnessConfig(HarnessConfig):
89
- version: str = "0.80.10"
91
+ version: str = Field(default="0.84.0", pattern=r"^[A-Za-z0-9._+-]+$")
90
92
  """Pi release to install, pinned for reproducibility."""
91
93
 
92
94
 
@@ -223,4 +225,6 @@ class PiHarness(Harness[PiHarnessConfig]):
223
225
  ACP_COMMAND,
224
226
  prompt,
225
227
  session_path=f"{agent_dir}/acp-session",
228
+ # Pi can end after its final tool completes without a text message.
229
+ allow_empty_tool_reply=True,
226
230
  )
@@ -31,7 +31,7 @@ POOL_ACP = ACP()
31
31
 
32
32
 
33
33
  class PoolHarnessConfig(HarnessConfig):
34
- version: str = Field(default="1.0.11", pattern=r"^[A-Za-z0-9._+-]+$")
34
+ version: str = Field(default="1.0.15", pattern=r"^[A-Za-z0-9._+-]+$")
35
35
  """Pool release to install, pinned for reproducibility."""
36
36
 
37
37
 
@@ -30,7 +30,7 @@ RLM_ACP = ACP()
30
30
 
31
31
 
32
32
  class RLMHarnessConfig(HarnessConfig):
33
- version: str = "main"
33
+ version: str = Field(default="main", min_length=1)
34
34
  """Git ref (branch, tag, or commit) of rlm-harness to install."""
35
35
  max_depth: int = 0
36
36
  """Recursion depth rlm may spawn sub-harnesses to (RLM_MAX_DEPTH)."""
@@ -1,6 +1,8 @@
1
1
  import logging
2
2
  from pathlib import Path
3
3
 
4
+ from pydantic import Field
5
+
4
6
  from verifiers.v1.clients import ModelContext
5
7
  from verifiers.v1.configs.harness import HarnessConfig
6
8
  from verifiers.v1.harness import Harness
@@ -13,7 +15,7 @@ logger = logging.getLogger(__name__)
13
15
 
14
16
 
15
17
  class Terminus2HarnessConfig(HarnessConfig):
16
- version: str = "0.20.0"
18
+ version: str = Field(default="0.20.0", pattern=r"^[A-Za-z0-9._+-]+$")
17
19
  """Harbor release to install, pinned for reproducibility."""
18
20
 
19
21
 
@@ -9,7 +9,9 @@ SSE requests are supported.
9
9
  One server multiplexes many rollouts: each rollout registers separate model and state
10
10
  capabilities, and the server routes each to the right session. So N rollouts need one
11
11
  server (and, behind a remote runtime, one tunnel) per pool member rather than one each —
12
- see `interception.pool`.
12
+ see `interception.pool`. The server also owns the model clients (one per distinct endpoint
13
+ config, assigned to each session at register and closed with the server), so its rollouts
14
+ share one bounded keepalive connection pool upstream instead of churning per-rollout TCP.
13
15
 
14
16
  The server is a pure model boundary: one request, one turn — refusal checks (limits,
15
17
  `@stop`s), the model call, the graph commit, retry atomicity. A run's user exchange
@@ -34,6 +36,8 @@ from pydantic import ValidationError
34
36
  from pydantic_core import PydanticSerializationError, from_json, to_json
35
37
 
36
38
  from verifiers.v1 import graph
39
+ from verifiers.v1.clients import Client, resolve_client
40
+ from verifiers.v1.configs.client import BaseClientConfig
37
41
  from verifiers.v1.dialects import DIALECTS, Dialect
38
42
  from verifiers.v1.dialects.base import is_sse_done_event
39
43
  from verifiers.v1.errors import (
@@ -130,6 +134,7 @@ class InterceptionServer(Interception):
130
134
  ) -> None:
131
135
  super().__init__()
132
136
  self.sessions: dict[str, RolloutSession] = {}
137
+ self.clients: dict[str, Client] = {}
133
138
  self.state_sessions: dict[str, RolloutSession] = {}
134
139
  self.state_routes: dict[str, RolloutSession] = {}
135
140
  self.state_service_secrets = frozenset(state_service_secrets)
@@ -147,8 +152,22 @@ class InterceptionServer(Interception):
147
152
  """Rollouts currently registered — what the pools balance on."""
148
153
  return len(self.sessions)
149
154
 
155
+ def _client(self, config: BaseClientConfig) -> Client:
156
+ """The server-owned client for `config` — one per distinct endpoint config, shared
157
+ by every session registered under it, so the rollouts this server multiplexes reuse
158
+ one bounded keepalive pool instead of each opening (and tearing down) their own
159
+ connections. Closed with the server."""
160
+ key = config.model_dump_json()
161
+ client = self.clients.get(key)
162
+ if client is None:
163
+ client = self.clients[key] = resolve_client(config)
164
+ self.stack.push_async_callback(client.close)
165
+ return client
166
+
150
167
  def register(self, session: RolloutSession) -> tuple[str, str]:
151
- """Register separate capabilities for model inference and private task state."""
168
+ """Register separate capabilities for model inference and private task state, and
169
+ assign the session its server-owned model client."""
170
+ session.client = self._client(session.ctx.client)
152
171
  model_secret = secrets.token_urlsafe(16)
153
172
  state_secret = secrets.token_urlsafe(16)
154
173
  self.sessions[model_secret] = session