verifiers 0.2.2.dev55__py3-none-any.whl → 0.2.2.dev57__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/agent.py +5 -2
- verifiers/v1/cli/init.py +7 -2
- verifiers/v1/env.py +5 -12
- verifiers/v1/harnesses/__init__.py +6 -0
- verifiers/v1/harnesses/browser_use/__init__.py +6 -0
- verifiers/v1/harnesses/browser_use/harness.py +125 -0
- verifiers/v1/harnesses/browser_use/program.py +402 -0
- verifiers/v1/mcp/launch.py +2 -2
- verifiers/v1/rollout.py +3 -3
- verifiers/v1/task.py +11 -37
- verifiers/v1/taskset.py +13 -17
- verifiers/v1/utils/compile.py +10 -9
- {verifiers-0.2.2.dev55.dist-info → verifiers-0.2.2.dev57.dist-info}/METADATA +1 -1
- {verifiers-0.2.2.dev55.dist-info → verifiers-0.2.2.dev57.dist-info}/RECORD +17 -14
- {verifiers-0.2.2.dev55.dist-info → verifiers-0.2.2.dev57.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev55.dist-info → verifiers-0.2.2.dev57.dist-info}/entry_points.txt +0 -0
- {verifiers-0.2.2.dev55.dist-info → verifiers-0.2.2.dev57.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/agent.py
CHANGED
|
@@ -335,7 +335,7 @@ class Agent:
|
|
|
335
335
|
if self._server is None:
|
|
336
336
|
return None
|
|
337
337
|
if self._server.tunnel is not None or (
|
|
338
|
-
run_is_local and not shared_tools and not
|
|
338
|
+
run_is_local and not shared_tools and not task.toolsets(task.config)
|
|
339
339
|
):
|
|
340
340
|
return self._server
|
|
341
341
|
return None
|
|
@@ -517,7 +517,10 @@ class Agent:
|
|
|
517
517
|
)
|
|
518
518
|
run_is_local = runtime_is_local(runtime_config)
|
|
519
519
|
validate_pairing(
|
|
520
|
-
self.harness,
|
|
520
|
+
self.harness,
|
|
521
|
+
type(task),
|
|
522
|
+
runtime_config,
|
|
523
|
+
tools=[*task.toolsets(task.config), *shared_tools.values()],
|
|
521
524
|
)
|
|
522
525
|
# Timeout precedence: agent-level wins, else the task's, else no limit.
|
|
523
526
|
agent_timeout = (
|
verifiers/v1/cli/init.py
CHANGED
|
@@ -71,7 +71,12 @@ def _taskset_py(pkg: str, prefix: str, *, add_tool: bool) -> str:
|
|
|
71
71
|
if add_tool:
|
|
72
72
|
local_imports.append(f"from {pkg}.servers.tool import {prefix}Toolset")
|
|
73
73
|
task_config_fields += "\n tools: vf.ToolsetConfig = vf.ToolsetConfig()"
|
|
74
|
-
task_decls +=
|
|
74
|
+
task_decls += (
|
|
75
|
+
"\n\n @classmethod"
|
|
76
|
+
f"\n def toolsets(cls, config: {prefix}TaskConfig) -> list[vf.Toolset]:"
|
|
77
|
+
f"\n return [{prefix}Toolset(config.tools)]"
|
|
78
|
+
"\n"
|
|
79
|
+
)
|
|
75
80
|
if local_imports:
|
|
76
81
|
imports += "\n\n" + "\n".join(local_imports)
|
|
77
82
|
methods_block = ""
|
|
@@ -178,7 +183,7 @@ def _readme(dash: str, pkg: str, *, add_tool: bool, add_harness: bool) -> str:
|
|
|
178
183
|
]
|
|
179
184
|
if add_tool:
|
|
180
185
|
layout.append(
|
|
181
|
-
f"- `{pkg}/servers/tool.py` — a `vf.Toolset` tool server,
|
|
186
|
+
f"- `{pkg}/servers/tool.py` — a `vf.Toolset` tool server, constructed in `Task.toolsets`."
|
|
182
187
|
)
|
|
183
188
|
if add_harness:
|
|
184
189
|
layout.append(
|
verifiers/v1/env.py
CHANGED
|
@@ -31,7 +31,7 @@ from verifiers.v1.interception import (
|
|
|
31
31
|
from verifiers.v1.mcp import SharedToolServer, serve_shared
|
|
32
32
|
from verifiers.v1.retries import run_episode_with_retry
|
|
33
33
|
from verifiers.v1.runtimes import SubprocessConfig, runtime_is_local
|
|
34
|
-
from verifiers.v1.task import Task
|
|
34
|
+
from verifiers.v1.task import Task
|
|
35
35
|
from verifiers.v1.trace import Error, Trace
|
|
36
36
|
from verifiers.v1.utils.generic import concrete_type
|
|
37
37
|
from verifiers.v1.utils.memory import trim_memory_periodically
|
|
@@ -378,24 +378,17 @@ class Env(ABC, Generic[ConfigT]):
|
|
|
378
378
|
|
|
379
379
|
def _requires_tunnel(self, shared: dict[str, SharedToolServer]) -> bool:
|
|
380
380
|
"""`requires_tunnel` over the consumers known before any rollout: role
|
|
381
|
-
runtimes, live `shared` servers, and the task class's tool servers
|
|
382
|
-
|
|
381
|
+
runtimes, live `shared` servers, and the task class's tool servers
|
|
382
|
+
(their configs read off the declared `tools` fields, no task needed)."""
|
|
383
383
|
task_cls = type(self.taskset).task_type()
|
|
384
|
-
server_classes = [*task_cls.tools]
|
|
385
|
-
if server_classes and task_cls.server_config is not Task.server_config:
|
|
386
|
-
return True
|
|
387
|
-
sole = len({*task_cls.tools}) == 1
|
|
388
384
|
configs = [
|
|
389
|
-
|
|
390
|
-
task_cls.__name__, self.taskset.config.task, server_cls, sole=sole
|
|
391
|
-
)
|
|
392
|
-
for server_cls in server_classes
|
|
385
|
+
server.config for server in task_cls.toolsets(self.taskset.config.task)
|
|
393
386
|
]
|
|
394
387
|
return requires_tunnel(self._runs_local(), configs, shared.values())
|
|
395
388
|
|
|
396
389
|
@contextlib.asynccontextmanager
|
|
397
390
|
async def shared_tools(self):
|
|
398
|
-
servers = self.taskset.
|
|
391
|
+
servers = self.taskset.toolsets(self.taskset.config)
|
|
399
392
|
if not servers:
|
|
400
393
|
yield {}
|
|
401
394
|
return
|
|
@@ -1,4 +1,8 @@
|
|
|
1
1
|
from verifiers.v1.harnesses.bash import BashHarness, BashHarnessConfig
|
|
2
|
+
from verifiers.v1.harnesses.browser_use import (
|
|
3
|
+
BrowserUseHarness,
|
|
4
|
+
BrowserUseHarnessConfig,
|
|
5
|
+
)
|
|
2
6
|
from verifiers.v1.harnesses.claude_code import (
|
|
3
7
|
ClaudeCodeHarness,
|
|
4
8
|
ClaudeCodeHarnessConfig,
|
|
@@ -18,6 +22,8 @@ from verifiers.v1.harnesses.terminus_2 import Terminus2Harness, Terminus2Harness
|
|
|
18
22
|
__all__ = [
|
|
19
23
|
"BashHarness",
|
|
20
24
|
"BashHarnessConfig",
|
|
25
|
+
"BrowserUseHarness",
|
|
26
|
+
"BrowserUseHarnessConfig",
|
|
21
27
|
"ClaudeCodeHarness",
|
|
22
28
|
"ClaudeCodeHarnessConfig",
|
|
23
29
|
"CodexHarness",
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
import json
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from typing import Literal
|
|
4
|
+
|
|
5
|
+
from pydantic import model_validator
|
|
6
|
+
|
|
7
|
+
from verifiers.v1.clients import ModelContext
|
|
8
|
+
from verifiers.v1.configs.harness import HarnessConfig
|
|
9
|
+
from verifiers.v1.dialects.chat import message_to_wire
|
|
10
|
+
from verifiers.v1.harness import Harness
|
|
11
|
+
from verifiers.v1.runtimes import ProgramResult, Runtime
|
|
12
|
+
from verifiers.v1.task import TaskData
|
|
13
|
+
from verifiers.v1.trace import Trace
|
|
14
|
+
|
|
15
|
+
PROGRAM_SOURCE = (Path(__file__).resolve().parent / "program.py").read_text()
|
|
16
|
+
|
|
17
|
+
# The helper names and persistence rules the model needs to use the local tool.
|
|
18
|
+
BROWSER_SYSTEM_PROMPT = """You are a browser automation agent. Your `browser` tool executes Python code that controls a real Chromium over CDP through browser-harness; its helpers are pre-imported.
|
|
19
|
+
|
|
20
|
+
Rules of the tool:
|
|
21
|
+
- Each call runs in a fresh Python process: variables do NOT persist between calls. The browser does persist — tabs, cookies, and page state carry over.
|
|
22
|
+
- Use print() for anything you want to see. Filter in Python before printing; a raw AX tree or DOM dump is huge.
|
|
23
|
+
- The first navigation is new_tab(url), not goto_url(url). After navigating, call wait_for_load().
|
|
24
|
+
|
|
25
|
+
Core helpers: new_tab(url), goto_url(url), page_info(), js(expression), click_at_xy(x, y), type_text(text), press_key(key), fill_input(selector, text), scroll(x, y, dy), wait_for_load(), wait_for_element(selector), wait_for_network_idle(), list_tabs(), switch_tab(target), close_tab(), ensure_real_tab(), capture_screenshot(path), upload_file(selector, path), and raw cdp("Domain.method", ...).
|
|
26
|
+
|
|
27
|
+
Finding elements: prefer the accessibility tree over screenshots. cdp("Accessibility.getFullAXTree")["nodes"] has every element's role, name, and backendDOMNodeId — filter in Python before printing. For coordinates: q = cdp("DOM.getBoxModel", backendNodeId=n)["model"]["content"]; x, y = sum(q[0::2])/4, sum(q[1::2])/4, then click_at_xy(x, y) and verify with a targeted js(...) or page_info() check. Fall back to js(...) over the DOM when the AX tree lacks the element."""
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class BrowserUseHarnessConfig(HarnessConfig):
|
|
31
|
+
browser: Literal["chromium", "cdp"] = "chromium"
|
|
32
|
+
"""`chromium` launches locally; `cdp` attaches to `cdp_url` without owning it."""
|
|
33
|
+
|
|
34
|
+
cdp_url: str | None = None
|
|
35
|
+
"""Rollout-scoped HTTP or WebSocket endpoint used with `browser = "cdp"`."""
|
|
36
|
+
|
|
37
|
+
@model_validator(mode="after")
|
|
38
|
+
def _require_cdp_url_iff_cdp(self) -> "BrowserUseHarnessConfig":
|
|
39
|
+
if self.browser == "cdp" and not self.cdp_url:
|
|
40
|
+
raise ValueError(
|
|
41
|
+
"browser='cdp' needs cdp_url set to a CDP endpoint to attach to"
|
|
42
|
+
)
|
|
43
|
+
if self.browser != "cdp" and self.cdp_url:
|
|
44
|
+
raise ValueError(
|
|
45
|
+
f"cdp_url is only valid with browser='cdp', not browser={self.browser!r}"
|
|
46
|
+
)
|
|
47
|
+
return self
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class BrowserUseHarness(Harness[BrowserUseHarnessConfig]):
|
|
51
|
+
APPENDS_SYSTEM_PROMPT = True
|
|
52
|
+
SUPPORTS_MCP = True
|
|
53
|
+
SUPPORTS_RESUME = True
|
|
54
|
+
# The browser tool executes model-authored Python through a third-party daemon.
|
|
55
|
+
NEEDS_CONTAINER = True
|
|
56
|
+
|
|
57
|
+
async def setup(self, runtime: Runtime) -> None:
|
|
58
|
+
await runtime.prepare_uv_script(PROGRAM_SOURCE, self.config.resolved_env)
|
|
59
|
+
|
|
60
|
+
async def launch(
|
|
61
|
+
self,
|
|
62
|
+
ctx: ModelContext,
|
|
63
|
+
trace: Trace,
|
|
64
|
+
runtime: Runtime,
|
|
65
|
+
endpoint: str,
|
|
66
|
+
secret: str,
|
|
67
|
+
mcp_urls: dict[str, str],
|
|
68
|
+
data: TaskData,
|
|
69
|
+
) -> ProgramResult:
|
|
70
|
+
system_prompt, prompt = self.resolve_prompt(data)
|
|
71
|
+
system_prompt = "\n\n".join(
|
|
72
|
+
p for p in (BROWSER_SYSTEM_PROMPT, system_prompt) if p
|
|
73
|
+
)
|
|
74
|
+
# Default resume replays the transcript. If there was no task system
|
|
75
|
+
# prompt, that transcript already contains this harness prompt.
|
|
76
|
+
replaying_browser_prompt = (
|
|
77
|
+
data.system_prompt is None
|
|
78
|
+
and prompt is not None
|
|
79
|
+
and not isinstance(prompt, str)
|
|
80
|
+
and any(
|
|
81
|
+
message.role == "system" and message.content == BROWSER_SYSTEM_PROMPT
|
|
82
|
+
for message in prompt
|
|
83
|
+
)
|
|
84
|
+
)
|
|
85
|
+
env = {**self.config.resolved_env}
|
|
86
|
+
state = f".vf-browser-{trace.id}"
|
|
87
|
+
args = [
|
|
88
|
+
f"--base-url={endpoint}",
|
|
89
|
+
f"--api-key={secret}",
|
|
90
|
+
f"--model={ctx.model}",
|
|
91
|
+
f"--browser={self.config.browser}",
|
|
92
|
+
# A resumed segment reuses this trace's browser and profile.
|
|
93
|
+
f"--state-dir={state}",
|
|
94
|
+
]
|
|
95
|
+
if not replaying_browser_prompt:
|
|
96
|
+
args.append(f"--system-prompt={system_prompt}")
|
|
97
|
+
if self.config.cdp_url:
|
|
98
|
+
args.append(f"--cdp-url={self.config.cdp_url}")
|
|
99
|
+
if mcp_urls:
|
|
100
|
+
# The program connects to the tool servers over HTTP; hand it a standard
|
|
101
|
+
# `mcpServers` URL config (the `mcp` client itself comes from the uv deps).
|
|
102
|
+
args.append(
|
|
103
|
+
"--mcp-config="
|
|
104
|
+
+ json.dumps(
|
|
105
|
+
{
|
|
106
|
+
"mcpServers": {
|
|
107
|
+
name: {"url": url} for name, url in mcp_urls.items()
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
)
|
|
111
|
+
)
|
|
112
|
+
if isinstance(prompt, str):
|
|
113
|
+
args.append(f"--prompt={prompt}")
|
|
114
|
+
elif prompt is not None:
|
|
115
|
+
# Base64 images can exceed exec limits, so hand Messages off through a file.
|
|
116
|
+
path = f".vf-initial-messages-{trace.id}.json"
|
|
117
|
+
await runtime.write(
|
|
118
|
+
path,
|
|
119
|
+
json.dumps([message_to_wire(m) for m in prompt]).encode(),
|
|
120
|
+
)
|
|
121
|
+
args.append(f"--initial-messages-file={path}")
|
|
122
|
+
program = await runtime.prepare_uv_script(
|
|
123
|
+
PROGRAM_SOURCE, self.config.resolved_env
|
|
124
|
+
)
|
|
125
|
+
return await runtime.run_program([*program, *args], env)
|
|
@@ -0,0 +1,402 @@
|
|
|
1
|
+
# /// script
|
|
2
|
+
# requires-python = ">=3.10"
|
|
3
|
+
# dependencies = [
|
|
4
|
+
# "browser-harness==0.1.8",
|
|
5
|
+
# "openai",
|
|
6
|
+
# "mcp>=1.24.0,<2",
|
|
7
|
+
# "httpx",
|
|
8
|
+
# "tenacity",
|
|
9
|
+
# ]
|
|
10
|
+
# ///
|
|
11
|
+
"""A chat loop whose one local tool drives a real Chromium over CDP.
|
|
12
|
+
|
|
13
|
+
Each tool call pipes the model's code to browser-harness, whose daemon holds the
|
|
14
|
+
CDP connection and pre-imports its page helpers. `chromium` launches and owns a
|
|
15
|
+
local browser; `cdp` attaches to an HTTP or WebSocket endpoint it does not own.
|
|
16
|
+
Trace-scoped state preserves tabs across resume and records owned process IDs.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
import argparse
|
|
20
|
+
import asyncio
|
|
21
|
+
import json
|
|
22
|
+
import os
|
|
23
|
+
import re
|
|
24
|
+
import shutil
|
|
25
|
+
import subprocess
|
|
26
|
+
import sys
|
|
27
|
+
import time
|
|
28
|
+
import urllib.request
|
|
29
|
+
from contextlib import AsyncExitStack, asynccontextmanager, suppress
|
|
30
|
+
from pathlib import Path
|
|
31
|
+
from typing import Any, cast
|
|
32
|
+
from urllib.parse import urlsplit
|
|
33
|
+
|
|
34
|
+
import httpx
|
|
35
|
+
from openai import AsyncOpenAI
|
|
36
|
+
from tenacity import AsyncRetrying, stop_after_attempt, wait_exponential_jitter
|
|
37
|
+
|
|
38
|
+
MCP_CALL_ATTEMPTS = 6
|
|
39
|
+
MCP_TIMEOUT = httpx.Timeout(600.0, connect=5.0) # the OpenAI SDK client defaults
|
|
40
|
+
|
|
41
|
+
BROWSER_TOOL_TIMEOUT = 3600
|
|
42
|
+
"""Matches the bash harness's command timeout."""
|
|
43
|
+
|
|
44
|
+
BROWSER_READY_TIMEOUT = 60
|
|
45
|
+
"""The documented Playwright image announced CDP in under one second from a
|
|
46
|
+
cold container; one minute leaves ample runtime startup headroom."""
|
|
47
|
+
|
|
48
|
+
_DEVTOOLS_LINE = re.compile(r"DevTools listening on ws://127\.0\.0\.1:(\d+)/")
|
|
49
|
+
|
|
50
|
+
BROWSER_TOOL = {
|
|
51
|
+
"type": "function",
|
|
52
|
+
"function": {
|
|
53
|
+
"name": "browser",
|
|
54
|
+
"description": (
|
|
55
|
+
"Execute Python code that drives the browser through browser-harness. "
|
|
56
|
+
"Helpers are pre-imported; use print() to see values. Each call runs in "
|
|
57
|
+
"a fresh process: Python variables do not persist between calls, but the "
|
|
58
|
+
"browser (tabs, cookies, page state) does."
|
|
59
|
+
),
|
|
60
|
+
"parameters": {
|
|
61
|
+
"type": "object",
|
|
62
|
+
"properties": {
|
|
63
|
+
"code": {
|
|
64
|
+
"type": "string",
|
|
65
|
+
"description": "The Python code to execute.",
|
|
66
|
+
}
|
|
67
|
+
},
|
|
68
|
+
"required": ["code"],
|
|
69
|
+
},
|
|
70
|
+
},
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def find_browser() -> str:
|
|
75
|
+
"""The Chromium binary to launch, where a machine keeps one.
|
|
76
|
+
|
|
77
|
+
`PLAYWRIGHT_BROWSERS_PATH` before PATH: an image that installs browsers
|
|
78
|
+
through Playwright puts nothing on PATH.
|
|
79
|
+
"""
|
|
80
|
+
for key in ("BH_CHROME_PATH", "CHROME_PATH"):
|
|
81
|
+
candidate = os.environ.get(key)
|
|
82
|
+
if candidate and os.access(candidate, os.X_OK):
|
|
83
|
+
return candidate
|
|
84
|
+
registry = os.environ.get("PLAYWRIGHT_BROWSERS_PATH")
|
|
85
|
+
if registry:
|
|
86
|
+
builds = sorted(Path(registry).glob("chromium-*/chrome-linux*/chrome"))
|
|
87
|
+
if builds:
|
|
88
|
+
return str(builds[-1])
|
|
89
|
+
for name in (
|
|
90
|
+
"google-chrome-stable",
|
|
91
|
+
"google-chrome",
|
|
92
|
+
"chromium",
|
|
93
|
+
"chromium-browser",
|
|
94
|
+
):
|
|
95
|
+
found = shutil.which(name)
|
|
96
|
+
if found:
|
|
97
|
+
return found
|
|
98
|
+
raise SystemExit(
|
|
99
|
+
"no Chromium/Chrome found; run on a browser-capable image "
|
|
100
|
+
"(e.g. mcr.microsoft.com/playwright/python) or set BH_CHROME_PATH"
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _endpoint_alive(endpoint: str) -> bool:
|
|
105
|
+
try:
|
|
106
|
+
# browser-harness uses five seconds for this same DevTools HTTP probe.
|
|
107
|
+
urllib.request.urlopen(f"{endpoint}/json/version", timeout=5).close()
|
|
108
|
+
return True
|
|
109
|
+
except OSError:
|
|
110
|
+
return False
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def ensure_chromium(state_dir: Path) -> str:
|
|
114
|
+
"""Reuse this trace's live Chromium or launch it."""
|
|
115
|
+
endpoint_file = state_dir / "cdp-endpoint"
|
|
116
|
+
if endpoint_file.exists():
|
|
117
|
+
endpoint = endpoint_file.read_text().strip()
|
|
118
|
+
if endpoint and _endpoint_alive(endpoint):
|
|
119
|
+
return endpoint
|
|
120
|
+
|
|
121
|
+
profile = state_dir / "profile"
|
|
122
|
+
profile.mkdir(parents=True, exist_ok=True)
|
|
123
|
+
log = state_dir / "browser.log"
|
|
124
|
+
with open(log, "wb") as stderr:
|
|
125
|
+
browser = subprocess.Popen(
|
|
126
|
+
[
|
|
127
|
+
find_browser(),
|
|
128
|
+
"--remote-debugging-port=0",
|
|
129
|
+
f"--user-data-dir={profile}",
|
|
130
|
+
"--headless",
|
|
131
|
+
"--no-first-run",
|
|
132
|
+
"--no-default-browser-check",
|
|
133
|
+
"--disable-dev-shm-usage",
|
|
134
|
+
"--no-sandbox",
|
|
135
|
+
],
|
|
136
|
+
stdout=subprocess.DEVNULL,
|
|
137
|
+
stderr=stderr,
|
|
138
|
+
stdin=subprocess.DEVNULL,
|
|
139
|
+
start_new_session=True,
|
|
140
|
+
)
|
|
141
|
+
deadline = time.monotonic() + BROWSER_READY_TIMEOUT
|
|
142
|
+
while time.monotonic() < deadline and browser.poll() is None:
|
|
143
|
+
output = log.read_text(errors="replace")
|
|
144
|
+
if match := _DEVTOOLS_LINE.search(output):
|
|
145
|
+
endpoint = f"http://127.0.0.1:{match.group(1)}"
|
|
146
|
+
endpoint_file.write_text(endpoint)
|
|
147
|
+
return endpoint
|
|
148
|
+
time.sleep(0.1)
|
|
149
|
+
if browser.poll() is None:
|
|
150
|
+
browser.kill()
|
|
151
|
+
browser.wait()
|
|
152
|
+
tail = log.read_text(errors="replace")[-2000:]
|
|
153
|
+
raise SystemExit(
|
|
154
|
+
f"browser did not announce a DevTools port within {BROWSER_READY_TIMEOUT}s: "
|
|
155
|
+
f"{tail or '<no browser output>'}"
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def run_browser(code: str, env: dict[str, str]) -> str:
|
|
160
|
+
"""Run model code through the pinned browser-harness CLI."""
|
|
161
|
+
try:
|
|
162
|
+
result = subprocess.run(
|
|
163
|
+
[sys.executable, "-m", "browser_harness.run"],
|
|
164
|
+
input=code,
|
|
165
|
+
capture_output=True,
|
|
166
|
+
text=True,
|
|
167
|
+
timeout=BROWSER_TOOL_TIMEOUT,
|
|
168
|
+
env=env,
|
|
169
|
+
check=False,
|
|
170
|
+
)
|
|
171
|
+
return (result.stdout + result.stderr) or "(no output)"
|
|
172
|
+
except Exception as e: # noqa: BLE001 - tool failures are returned to the model
|
|
173
|
+
return f"error: {e}"
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
async def chat(
|
|
177
|
+
client: AsyncOpenAI,
|
|
178
|
+
model: str,
|
|
179
|
+
messages: list[dict[str, Any]],
|
|
180
|
+
tools: list[dict[str, Any]],
|
|
181
|
+
):
|
|
182
|
+
completion = await client.chat.completions.create(
|
|
183
|
+
model=model,
|
|
184
|
+
messages=cast(Any, messages),
|
|
185
|
+
tools=cast(Any, tools or None),
|
|
186
|
+
)
|
|
187
|
+
return completion.choices[0].message
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
@asynccontextmanager
|
|
191
|
+
async def mcp_session(spec: dict):
|
|
192
|
+
"""One fresh streamable-HTTP session to an MCP server, opened and closed within the caller's
|
|
193
|
+
task so AnyIO cancellation scopes stay correctly nested. A teardown failure after the body
|
|
194
|
+
completed is swallowed — the result is already in hand, and closing noise must not fail (or
|
|
195
|
+
replay) an already-answered call."""
|
|
196
|
+
from mcp import ClientSession
|
|
197
|
+
from mcp.client.streamable_http import (
|
|
198
|
+
create_mcp_http_client,
|
|
199
|
+
streamable_http_client,
|
|
200
|
+
)
|
|
201
|
+
|
|
202
|
+
stack = AsyncExitStack()
|
|
203
|
+
try:
|
|
204
|
+
http_client = await stack.enter_async_context(
|
|
205
|
+
create_mcp_http_client(
|
|
206
|
+
headers=spec.get("headers") or None, timeout=MCP_TIMEOUT
|
|
207
|
+
)
|
|
208
|
+
)
|
|
209
|
+
read, write, *_ = await stack.enter_async_context(
|
|
210
|
+
streamable_http_client(spec["url"], http_client=http_client)
|
|
211
|
+
)
|
|
212
|
+
session = await stack.enter_async_context(ClientSession(read, write))
|
|
213
|
+
await session.initialize()
|
|
214
|
+
yield session
|
|
215
|
+
finally:
|
|
216
|
+
with suppress(Exception):
|
|
217
|
+
await stack.aclose()
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
async def with_retry(call):
|
|
221
|
+
"""Run one session-scoped operation, retrying transient failures with backoff. A call whose
|
|
222
|
+
response was lost may be replayed — MCP has no idempotency key, so tools should tolerate
|
|
223
|
+
at-least-once delivery (a tool that fails reports through its result, not an exception)."""
|
|
224
|
+
async for attempt in AsyncRetrying(
|
|
225
|
+
stop=stop_after_attempt(MCP_CALL_ATTEMPTS),
|
|
226
|
+
wait=wait_exponential_jitter(initial=0.5, max=30),
|
|
227
|
+
reraise=True,
|
|
228
|
+
):
|
|
229
|
+
with attempt:
|
|
230
|
+
return await call()
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
async def connect_mcp(
|
|
234
|
+
config: dict, reserved: set[str]
|
|
235
|
+
) -> tuple[list[dict], dict, dict]:
|
|
236
|
+
"""Enumerate each configured MCP server's tools (a streamable-HTTP `url`); return (tool schemas,
|
|
237
|
+
dispatch mapping `<server>_<tool>` -> (server name, raw tool name), servers mapping name -> spec).
|
|
238
|
+
No session is held — a stateless-HTTP server is reconnected per call."""
|
|
239
|
+
tool_schemas: list[dict] = []
|
|
240
|
+
dispatch: dict[str, tuple] = {}
|
|
241
|
+
servers: dict[str, dict] = {}
|
|
242
|
+
for name, spec in config.get("mcpServers", {}).items():
|
|
243
|
+
servers[name] = spec
|
|
244
|
+
|
|
245
|
+
async def list_tools(spec: dict = spec):
|
|
246
|
+
async with mcp_session(spec) as session:
|
|
247
|
+
return (await session.list_tools()).tools
|
|
248
|
+
|
|
249
|
+
for tool in await with_retry(list_tools):
|
|
250
|
+
# A server named "" (TOOL_PREFIX = None) advertises its tools bare.
|
|
251
|
+
full = f"{name}_{tool.name}" if name else tool.name
|
|
252
|
+
if full in reserved or full in dispatch:
|
|
253
|
+
raise ValueError(
|
|
254
|
+
f"duplicate tool name {full!r}; keep MCP tool names qualified"
|
|
255
|
+
)
|
|
256
|
+
tool_schemas.append(
|
|
257
|
+
{
|
|
258
|
+
"type": "function",
|
|
259
|
+
"function": {
|
|
260
|
+
"name": full,
|
|
261
|
+
"description": tool.description or "",
|
|
262
|
+
"parameters": tool.inputSchema,
|
|
263
|
+
},
|
|
264
|
+
}
|
|
265
|
+
)
|
|
266
|
+
dispatch[full] = (name, tool.name)
|
|
267
|
+
return tool_schemas, dispatch, servers
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def mcp_content_to_chat_content(blocks) -> str | list[dict]:
|
|
271
|
+
parts = []
|
|
272
|
+
for block in blocks:
|
|
273
|
+
if block.type == "text":
|
|
274
|
+
parts.append({"type": "text", "text": block.text})
|
|
275
|
+
elif block.type == "image":
|
|
276
|
+
url = f"data:{block.mimeType};base64,{block.data}"
|
|
277
|
+
parts.append({"type": "image_url", "image_url": {"url": url}})
|
|
278
|
+
else:
|
|
279
|
+
parts.append({"type": "text", "text": str(block)})
|
|
280
|
+
if not parts:
|
|
281
|
+
return str(blocks)
|
|
282
|
+
if all(part["type"] == "text" for part in parts):
|
|
283
|
+
return "\n".join(str(part["text"]) for part in parts)
|
|
284
|
+
return parts
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
async def call_mcp(
|
|
288
|
+
servers: dict, dispatch: dict, name: str, arguments: dict
|
|
289
|
+
) -> str | list[dict]:
|
|
290
|
+
"""Call a tool on a fresh session per attempt — see `with_retry` for the replay semantics.
|
|
291
|
+
The result is converted outside the retry so a conversion failure fails once."""
|
|
292
|
+
server_name, raw = dispatch[name]
|
|
293
|
+
|
|
294
|
+
async def call():
|
|
295
|
+
async with mcp_session(servers[server_name]) as session:
|
|
296
|
+
return await session.call_tool(raw, arguments)
|
|
297
|
+
|
|
298
|
+
result = await with_retry(call)
|
|
299
|
+
return mcp_content_to_chat_content(result.content)
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def parse_args() -> argparse.Namespace:
|
|
303
|
+
parser = argparse.ArgumentParser()
|
|
304
|
+
parser.add_argument("--base-url", required=True)
|
|
305
|
+
parser.add_argument("--api-key", required=True)
|
|
306
|
+
parser.add_argument("--model", required=True)
|
|
307
|
+
parser.add_argument("--state-dir", required=True)
|
|
308
|
+
parser.add_argument("--browser", default="chromium")
|
|
309
|
+
parser.add_argument("--cdp-url", default="")
|
|
310
|
+
parser.add_argument("--system-prompt", default="")
|
|
311
|
+
parser.add_argument("--prompt", default="")
|
|
312
|
+
parser.add_argument("--initial-messages-file", default="")
|
|
313
|
+
parser.add_argument("--mcp-config", default="")
|
|
314
|
+
return parser.parse_args()
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
async def main() -> None:
|
|
318
|
+
args = parse_args()
|
|
319
|
+
initial = []
|
|
320
|
+
if args.initial_messages_file:
|
|
321
|
+
path = Path(args.initial_messages_file)
|
|
322
|
+
payload = path.read_bytes()
|
|
323
|
+
path.unlink()
|
|
324
|
+
initial = json.loads(payload)
|
|
325
|
+
state_dir = Path(args.state_dir)
|
|
326
|
+
state_dir.mkdir(parents=True, exist_ok=True)
|
|
327
|
+
endpoint = (
|
|
328
|
+
ensure_chromium(state_dir) if args.browser == "chromium" else args.cdp_url
|
|
329
|
+
)
|
|
330
|
+
browser_env = {
|
|
331
|
+
**os.environ,
|
|
332
|
+
"BH_HOME": str(state_dir / "bh-home"),
|
|
333
|
+
"BH_TELEMETRY": "0",
|
|
334
|
+
(
|
|
335
|
+
"BU_CDP_WS" if urlsplit(endpoint).scheme in {"ws", "wss"} else "BU_CDP_URL"
|
|
336
|
+
): endpoint,
|
|
337
|
+
}
|
|
338
|
+
client = AsyncOpenAI(base_url=args.base_url, api_key=args.api_key)
|
|
339
|
+
config = json.loads(args.mcp_config or "{}")
|
|
340
|
+
tools = [BROWSER_TOOL]
|
|
341
|
+
reserved = {"browser"}
|
|
342
|
+
mcp_tools, dispatch, servers = (
|
|
343
|
+
await connect_mcp(config, reserved)
|
|
344
|
+
if config.get("mcpServers")
|
|
345
|
+
else ([], {}, {})
|
|
346
|
+
)
|
|
347
|
+
tools += mcp_tools
|
|
348
|
+
messages = (
|
|
349
|
+
[{"role": "system", "content": args.system_prompt}]
|
|
350
|
+
if args.system_prompt
|
|
351
|
+
else []
|
|
352
|
+
)
|
|
353
|
+
if initial:
|
|
354
|
+
messages.extend(initial)
|
|
355
|
+
elif args.prompt:
|
|
356
|
+
messages.append({"role": "user", "content": args.prompt})
|
|
357
|
+
while True:
|
|
358
|
+
message = await chat(client, args.model, messages, tools)
|
|
359
|
+
messages.append(message.model_dump(exclude_none=True))
|
|
360
|
+
if not message.tool_calls:
|
|
361
|
+
break
|
|
362
|
+
for call in message.tool_calls:
|
|
363
|
+
name = call.function.name
|
|
364
|
+
try:
|
|
365
|
+
tool_args = json.loads(call.function.arguments or "{}")
|
|
366
|
+
except json.JSONDecodeError as e:
|
|
367
|
+
messages.append(
|
|
368
|
+
{
|
|
369
|
+
"role": "tool",
|
|
370
|
+
"tool_call_id": call.id,
|
|
371
|
+
"content": f"error: invalid JSON in tool arguments ({e}); resend the call with valid JSON",
|
|
372
|
+
}
|
|
373
|
+
)
|
|
374
|
+
continue
|
|
375
|
+
# Valid JSON can still be a non-object (`[]`, `42`, `null`); the `.get(...)` calls
|
|
376
|
+
# below assume a dict, so reject anything else as a tool error rather than crashing.
|
|
377
|
+
if not isinstance(tool_args, dict):
|
|
378
|
+
messages.append(
|
|
379
|
+
{
|
|
380
|
+
"role": "tool",
|
|
381
|
+
"tool_call_id": call.id,
|
|
382
|
+
"content": f"error: tool arguments must be a JSON object, got {type(tool_args).__name__}; resend as an object",
|
|
383
|
+
}
|
|
384
|
+
)
|
|
385
|
+
continue
|
|
386
|
+
if name in dispatch:
|
|
387
|
+
content = await call_mcp(servers, dispatch, name, tool_args)
|
|
388
|
+
elif name == "browser":
|
|
389
|
+
content = await asyncio.to_thread(
|
|
390
|
+
run_browser,
|
|
391
|
+
tool_args.get("code", ""),
|
|
392
|
+
browser_env,
|
|
393
|
+
)
|
|
394
|
+
else:
|
|
395
|
+
content = f"error: unknown tool {name!r}"
|
|
396
|
+
messages.append(
|
|
397
|
+
{"role": "tool", "tool_call_id": call.id, "content": content}
|
|
398
|
+
)
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
if __name__ == "__main__":
|
|
402
|
+
asyncio.run(main())
|
verifiers/v1/mcp/launch.py
CHANGED
|
@@ -343,7 +343,7 @@ async def serve_shared(toolsets: list[Toolset], harness_is_local: bool = True):
|
|
|
343
343
|
name = toolset.server_name
|
|
344
344
|
if name in servers:
|
|
345
345
|
raise ToolsetError(
|
|
346
|
-
f"duplicate shared tool server name '{name}' in Taskset.
|
|
346
|
+
f"duplicate shared tool server name '{name}' in Taskset.toolsets — "
|
|
347
347
|
f"give one a distinct TOOL_PREFIX"
|
|
348
348
|
)
|
|
349
349
|
if type(toolset).setup_task is not ServerBase.setup_task:
|
|
@@ -351,7 +351,7 @@ async def serve_shared(toolsets: list[Toolset], harness_is_local: bool = True):
|
|
|
351
351
|
"shared server %r overrides `setup_task`, but `setup_task` is NEVER "
|
|
352
352
|
"called for a taskset-scoped server (it's built once, task-agnostic) — "
|
|
353
353
|
"its per-task logic will not run. Move task-agnostic work into `setup`, "
|
|
354
|
-
"or
|
|
354
|
+
"or construct it in `Task.toolsets` to run it per-rollout.",
|
|
355
355
|
name,
|
|
356
356
|
)
|
|
357
357
|
if cfg.url: # already running remotely; nothing launched, nothing to bridge
|
verifiers/v1/rollout.py
CHANGED
|
@@ -258,7 +258,7 @@ class RolloutRun:
|
|
|
258
258
|
):
|
|
259
259
|
await self.harness.setup(runtime)
|
|
260
260
|
async with boundary(ToolsetError, "building tool servers"):
|
|
261
|
-
|
|
261
|
+
toolsets = self.task.toolsets(self.task.config)
|
|
262
262
|
# `base_url` is the interception server's reachable URL for this rollout.
|
|
263
263
|
# The harness reaches the model at `{base_url}/v1`; tool servers reach this
|
|
264
264
|
# rollout's `/state` + `/task` at `base_url` — it's universally reachable
|
|
@@ -268,7 +268,7 @@ class RolloutRun:
|
|
|
268
268
|
self._interception,
|
|
269
269
|
runtime,
|
|
270
270
|
self._session,
|
|
271
|
-
|
|
271
|
+
toolsets,
|
|
272
272
|
self._shared_tools,
|
|
273
273
|
)
|
|
274
274
|
)
|
|
@@ -276,7 +276,7 @@ class RolloutRun:
|
|
|
276
276
|
self._secret = secret
|
|
277
277
|
self._urls = await self._stack.enter_async_context(
|
|
278
278
|
serve_tools(
|
|
279
|
-
|
|
279
|
+
toolsets,
|
|
280
280
|
runtime,
|
|
281
281
|
shared=self._shared_tools,
|
|
282
282
|
state_secret=secret,
|
verifiers/v1/task.py
CHANGED
|
@@ -16,7 +16,6 @@ from collections.abc import Mapping
|
|
|
16
16
|
from typing import TYPE_CHECKING, ClassVar, Generic, Self
|
|
17
17
|
|
|
18
18
|
from pydantic import BaseModel, ConfigDict, Field
|
|
19
|
-
from pydantic_config import BaseConfig
|
|
20
19
|
from typing_extensions import TypeVar
|
|
21
20
|
|
|
22
21
|
from verifiers.v1.artifacts import Artifact
|
|
@@ -122,37 +121,10 @@ DataT = TypeVar("DataT", bound=TaskData)
|
|
|
122
121
|
ConfigT = TypeVar("ConfigT", bound=TaskConfig, default=TaskConfig)
|
|
123
122
|
|
|
124
123
|
|
|
125
|
-
def resolve_server_config(
|
|
126
|
-
owner: str, config: BaseConfig, server_cls: type, *, sole: bool = True
|
|
127
|
-
) -> BaseConfig:
|
|
128
|
-
"""The config a declared server class is built with, resolved off `config`'s
|
|
129
|
-
fields: exact type match, else — only for a `sole` declared server — the unique
|
|
130
|
-
isinstance match, else a default-constructed one. Two matching fields raise; the
|
|
131
|
-
`server_config` methods are the explicit-pairing override. The isinstance
|
|
132
|
-
fallback is `sole`-gated because with several servers a subclass-typed field
|
|
133
|
-
could silently pair with the wrong one."""
|
|
134
|
-
cfg_cls = server_cls._config_cls()
|
|
135
|
-
values = {name: getattr(config, name) for name in type(config).model_fields}
|
|
136
|
-
matched = [name for name, v in values.items() if type(v) is cfg_cls]
|
|
137
|
-
if not matched and sole:
|
|
138
|
-
matched = [name for name, v in values.items() if isinstance(v, cfg_cls)]
|
|
139
|
-
if len(matched) > 1:
|
|
140
|
-
raise TaskError(
|
|
141
|
-
f"{owner}: ambiguous config for {server_cls.__name__} — config fields "
|
|
142
|
-
f"{matched} all match {cfg_cls.__name__}; override `server_config` to pair "
|
|
143
|
-
f"them explicitly"
|
|
144
|
-
)
|
|
145
|
-
if matched:
|
|
146
|
-
return values[matched[0]]
|
|
147
|
-
return cfg_cls()
|
|
148
|
-
|
|
149
|
-
|
|
150
124
|
class Task(Generic[DataT, StateT, ConfigT]):
|
|
151
125
|
NEEDS_CONTAINER: ClassVar[bool] = False
|
|
152
126
|
"""Whether the task needs a containerized environment (isolated filesystem, ...)."""
|
|
153
127
|
|
|
154
|
-
tools: ClassVar[tuple[type[Toolset], ...]] = ()
|
|
155
|
-
|
|
156
128
|
def __init__(self, data: DataT, config: ConfigT | None = None) -> None:
|
|
157
129
|
self.data = data
|
|
158
130
|
self.config = config if config is not None else self.config_type()()
|
|
@@ -249,16 +221,18 @@ class Task(Generic[DataT, StateT, ConfigT]):
|
|
|
249
221
|
|
|
250
222
|
return [load_judge(config) for config in self.config.judges]
|
|
251
223
|
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
224
|
+
@classmethod
|
|
225
|
+
def toolsets(cls, config: ConfigT) -> list[Toolset]:
|
|
226
|
+
"""Tool servers launched per rollout, each constructed with its config off
|
|
227
|
+
`config` — override and wire explicitly:
|
|
228
|
+
|
|
229
|
+
@classmethod
|
|
230
|
+
def toolsets(cls, config: MyTaskConfig) -> list[vf.Toolset]:
|
|
231
|
+
return [SearchToolset(config.tools)]
|
|
259
232
|
|
|
260
|
-
|
|
261
|
-
|
|
233
|
+
A classmethod so consumers can size placement (tunnels) off the class
|
|
234
|
+
before any task instance exists."""
|
|
235
|
+
return []
|
|
262
236
|
|
|
263
237
|
|
|
264
238
|
TaskT = TypeVar("TaskT", bound=Task)
|
verifiers/v1/taskset.py
CHANGED
|
@@ -19,13 +19,12 @@ import itertools
|
|
|
19
19
|
import random
|
|
20
20
|
from abc import ABC, abstractmethod
|
|
21
21
|
from collections.abc import Callable, Iterable, Iterator
|
|
22
|
-
from typing import TYPE_CHECKING,
|
|
22
|
+
from typing import TYPE_CHECKING, Generic, Self
|
|
23
23
|
|
|
24
|
-
from pydantic_config import BaseConfig
|
|
25
24
|
from typing_extensions import TypeVar
|
|
26
25
|
|
|
27
26
|
from verifiers.v1.configs.taskset import TasksetConfig
|
|
28
|
-
from verifiers.v1.task import Task, TaskT
|
|
27
|
+
from verifiers.v1.task import Task, TaskT
|
|
29
28
|
from verifiers.v1.utils.generic import concrete_type
|
|
30
29
|
from verifiers.v1.utils.sampling import SEED
|
|
31
30
|
|
|
@@ -40,10 +39,6 @@ class Taskset(ABC, Generic[TaskT, TasksetConfigT]):
|
|
|
40
39
|
"""Whether the taskset is infinite (yields tasks forever). Class-declared;
|
|
41
40
|
a `head(n)` view shadows it per instance (bounded by construction)."""
|
|
42
41
|
|
|
43
|
-
tools: ClassVar[tuple[type[Toolset], ...]] = ()
|
|
44
|
-
"""Tool servers shared by all tasks in the taskset. The environment will
|
|
45
|
-
spawn a single, global instance, reused across tasks."""
|
|
46
|
-
|
|
47
42
|
def __init__(self, config: TasksetConfigT) -> None:
|
|
48
43
|
self.config = config
|
|
49
44
|
override = config.system_prompt
|
|
@@ -102,13 +97,14 @@ class Taskset(ABC, Generic[TaskT, TasksetConfigT]):
|
|
|
102
97
|
def task_type(cls) -> type[Task]:
|
|
103
98
|
return concrete_type(cls, Task, origin=Taskset) or Task
|
|
104
99
|
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
100
|
+
@classmethod
|
|
101
|
+
def toolsets(cls, config: TasksetConfigT) -> list[Toolset]:
|
|
102
|
+
"""Tool servers shared by all tasks in the taskset (one global instance
|
|
103
|
+
per server, reused across an environment worker's rollouts), each
|
|
104
|
+
constructed with its config off `config` — override and wire explicitly:
|
|
105
|
+
|
|
106
|
+
@classmethod
|
|
107
|
+
def toolsets(cls, config: MyConfig) -> list[vf.Toolset]:
|
|
108
|
+
return [SearchToolset(config.tools)]
|
|
109
|
+
"""
|
|
110
|
+
return []
|
verifiers/v1/utils/compile.py
CHANGED
|
@@ -77,18 +77,19 @@ def validate_pairing(
|
|
|
77
77
|
task_cls: type[Task],
|
|
78
78
|
runtime_config: RuntimeConfig,
|
|
79
79
|
*,
|
|
80
|
-
|
|
80
|
+
tools: Collection = (),
|
|
81
81
|
) -> None:
|
|
82
82
|
"""Reject an impossible harness/task/runtime combination before any work happens.
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
if not harness.SUPPORTS_MCP and
|
|
83
|
+
A failure holds for every row the task class can carry. For `tools` only
|
|
84
|
+
emptiness matters — task-declared and shared servers alike mean MCP is in play.
|
|
85
|
+
(Hosting a user is interaction-scoped, not task-scoped — `Agent.interaction`
|
|
86
|
+
checks the harness can resume an exchange.)"""
|
|
87
|
+
if not harness.SUPPORTS_MCP and tools:
|
|
88
88
|
raise ValueError(
|
|
89
|
-
f"Harness {harness.config.id!r} does not support MCP tools, but "
|
|
90
|
-
f"{task_cls.__name__}
|
|
91
|
-
f"supports MCP (e.g. --env.agent.harness.id bash),
|
|
89
|
+
f"Harness {harness.config.id!r} does not support MCP tools, but the run "
|
|
90
|
+
f"serves some ({task_cls.__name__}'s or the taskset's shared servers). Run "
|
|
91
|
+
f"it with a harness that supports MCP (e.g. --env.agent.harness.id bash), "
|
|
92
|
+
f"or use tasks without tools."
|
|
92
93
|
)
|
|
93
94
|
if not harness.SUPPORTS_SKILLS and harness.config.skills:
|
|
94
95
|
raise ValueError(
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.dev57
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -168,10 +168,10 @@ verifiers/utils/tool_utils.py,sha256=gInWZODWQUZN4TyEMuuITyOUm9qlHJxtcvr1zZl_zJs
|
|
|
168
168
|
verifiers/utils/usage_utils.py,sha256=Mw6WC78r6EsLyRFdP9ICveqptkjizbSkAC86ZH0geVU,3918
|
|
169
169
|
verifiers/utils/version_utils.py,sha256=39gV8-mNF9GspAgKloLo8oy7_MVluMrwGPrntQv2hx4,2656
|
|
170
170
|
verifiers/v1/__init__.py,sha256=Im5qTZy9mH6PRTUi-1V6W7EWKZzJyTWLGKMYALbWz_w,7467
|
|
171
|
-
verifiers/v1/agent.py,sha256=
|
|
171
|
+
verifiers/v1/agent.py,sha256=t5Z_iI6Zyt0ieMDI7pfHUVYIKViuSXF-CI_rIApx3f4,32071
|
|
172
172
|
verifiers/v1/artifacts.py,sha256=5eKOjDSUWwj7GrhxILMHdz2ngEONWMdcTGlBcuGgrLU,5834
|
|
173
173
|
verifiers/v1/decorators.py,sha256=9CcCwgZlRjVGPqT5G1CFJUHP4OIMuBGkUJZkcI6dnbk,3865
|
|
174
|
-
verifiers/v1/env.py,sha256=
|
|
174
|
+
verifiers/v1/env.py,sha256=wJp9i49_LKVzMFXkwtt_SaRMUJ84i5r1t3QfNvnt4Ks,18178
|
|
175
175
|
verifiers/v1/episode.py,sha256=YtuRTV_PpwDglXFeN6UNg8IswpP1HseFLCYtascGMNI,3218
|
|
176
176
|
verifiers/v1/errors.py,sha256=slZnrtqMo_BgXQV46wJJhV6vvAAnVsuCbVGp2vBVWfs,6888
|
|
177
177
|
verifiers/v1/graph.py,sha256=aBAzh1Ibp3klKNYjvzNyJSbjV8Oi5TJ791dN7o6ex1E,29070
|
|
@@ -181,12 +181,12 @@ verifiers/v1/legacy.py,sha256=ISk3Bix9Hy5EjZtyV9QO0LNyDNDklBHVjFokOrw_3U8,22628
|
|
|
181
181
|
verifiers/v1/loaders.py,sha256=NSIBdNU3JVeyddL-dM4dJt6khCHyjCgzJuFtLrgiwD0,9989
|
|
182
182
|
verifiers/v1/push.py,sha256=MWZmvyj0iyzNIO0xRhlkCXtFF1plq6EJi8rWlvrFtvI,11363
|
|
183
183
|
verifiers/v1/retries.py,sha256=quUGVruUkZx-9WQcFiocMfL_M056voU8gwDqsuLR3f4,5412
|
|
184
|
-
verifiers/v1/rollout.py,sha256=
|
|
184
|
+
verifiers/v1/rollout.py,sha256=ehPoN6EnrIg_lvF1-pdf9pS8j47K3z6jDHc66MixzX0,20470
|
|
185
185
|
verifiers/v1/scoring.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
|
|
186
186
|
verifiers/v1/session.py,sha256=J6ZPuvsg6vMTJAs9dMUMtfdjCuvp6Hom877DVdKFKlY,6912
|
|
187
187
|
verifiers/v1/state.py,sha256=pcpN2V6tX7rIO5XdtdUYay1j0ktEwrMtq7w1ApfsPdE,675
|
|
188
|
-
verifiers/v1/task.py,sha256=
|
|
189
|
-
verifiers/v1/taskset.py,sha256=
|
|
188
|
+
verifiers/v1/task.py,sha256=skqOc-j2AxA6Wz2ikDD03o3cu1cVvr1G6gRJZrpOgfY,9059
|
|
189
|
+
verifiers/v1/taskset.py,sha256=s9T7v3REU1EKVCu2zORMA3gWinRK6LJk4GUvoKhAmEQ,4337
|
|
190
190
|
verifiers/v1/trace.py,sha256=_bGHOrE6fRLGm0JYc0PrrrO3lyr-qqOF5Y1bHfViwR0,19347
|
|
191
191
|
verifiers/v1/types.py,sha256=1PxamJspmoTc3OlFZAwH6o_2i_Tg9ZNvHxttIHWIxq8,8948
|
|
192
192
|
verifiers/v1/acp/__init__.py,sha256=9dwH6fLopNndRmcpRSYC8ghzzMquaLfgh645QM-Dwic,2189
|
|
@@ -194,7 +194,7 @@ verifiers/v1/acp/_runner.py,sha256=FqEnHR0Q_YrCt_EuaJqiqTpd2_kFa41FqcEunwOhvTI,6
|
|
|
194
194
|
verifiers/v1/cli/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
195
195
|
verifiers/v1/cli/debug.py,sha256=-jeEV_Q5tV8SOkWc1GdPZcXZWON_jU5gjeO4YoZm79o,11482
|
|
196
196
|
verifiers/v1/cli/gepa.py,sha256=hLIbS_4jBx0vvy4PXoePR5riFXcNvzQCMixZ-g2OR_U,4026
|
|
197
|
-
verifiers/v1/cli/init.py,sha256=
|
|
197
|
+
verifiers/v1/cli/init.py,sha256=Qj96Kwzfhaejhrc0yYqHNRl3ncoID028RGlHvS1C98Q,8508
|
|
198
198
|
verifiers/v1/cli/output.py,sha256=pMS5Z0EBgyAVShNCAk9jVURNeLOaTfeRWpt0J8hHjq8,5890
|
|
199
199
|
verifiers/v1/cli/replay.py,sha256=DAh9kB5PB-ppZLGhxdJwrCPch_Yky5HKhplwGFhKEz0,10164
|
|
200
200
|
verifiers/v1/cli/resolve.py,sha256=OsNw4C3r9xOVFa-mminJFcex_bAsJ6EQ8K7A_H8u2kQ,4370
|
|
@@ -252,10 +252,13 @@ verifiers/v1/gepa/config.py,sha256=gFn8Re4GoeHJs2PZ-iw4jCEZ72AiM1lNgwaA0uvJlYw,4
|
|
|
252
252
|
verifiers/v1/gepa/dataset.py,sha256=N2QYqUljHVuAdCbUuydEIlarvAWa6LFdGKu9qsOvjro,2186
|
|
253
253
|
verifiers/v1/gepa/reflection.py,sha256=ptHhx0lcDLcWhWhXuslh9T8gPVNpnX5XLPtXoneW1yA,1054
|
|
254
254
|
verifiers/v1/gepa/runner.py,sha256=-LZEt0lod6NKfyxg2dvOXZjfzTv3AWwendexd1De26c,6260
|
|
255
|
-
verifiers/v1/harnesses/__init__.py,sha256=
|
|
255
|
+
verifiers/v1/harnesses/__init__.py,sha256=706ES37vrh_kCbctnbgkXIb1jBtO2ozZY5klBSg8ss0,1460
|
|
256
256
|
verifiers/v1/harnesses/bash/__init__.py,sha256=IAV4eMMqDHSl-9ANb0Q3kxmBZ22SeHZFBM448X4gQio,140
|
|
257
257
|
verifiers/v1/harnesses/bash/harness.py,sha256=ZtjhsANuK3vXkf9mRCi8-sBMyPebJMBNHa58vTd2Fm8,5284
|
|
258
258
|
verifiers/v1/harnesses/bash/program.py,sha256=hGv3k9jwoabL045AGiZWiX76Cca3Juk58AaMYeJTEGo,15334
|
|
259
|
+
verifiers/v1/harnesses/browser_use/__init__.py,sha256=yxm2IG5TDLadxZEkX8SL26XWbcPcODXcrJfPfm1GqwU,171
|
|
260
|
+
verifiers/v1/harnesses/browser_use/harness.py,sha256=-P7CwMMkR7rLANAb6T4aOjAS4YuxzMPbm1MGvbIzbNI,5835
|
|
261
|
+
verifiers/v1/harnesses/browser_use/program.py,sha256=uk_nwTX_68bZckSFMkgtgwSkUK5HWBGqLTCDwxbt8P8,14287
|
|
259
262
|
verifiers/v1/harnesses/claude_code/__init__.py,sha256=JkZQylMSA3o1fvBx_NM41y_umuahmOSqFjTrJ2hryTw,171
|
|
260
263
|
verifiers/v1/harnesses/claude_code/harness.py,sha256=98k5T8d8nFQmxIU-ytkrkrrP6UNR5NxoGhVJRLVa-fI,4015
|
|
261
264
|
verifiers/v1/harnesses/codex/__init__.py,sha256=ocyBvlpSO8c3pT6D1ebrQohoXYESQcCyPT5kN3j1-N4,132
|
|
@@ -291,7 +294,7 @@ verifiers/v1/judges/reference.txt,sha256=Ej35kGXiT2uJ0LcHIeez49rCS_9uHmCQKoVdn93
|
|
|
291
294
|
verifiers/v1/judges/rubric.py,sha256=PglrJhMQR6scFyfz20jVlLQEzeqXKV2P6r0K0K7jjBE,11916
|
|
292
295
|
verifiers/v1/judges/rubric.txt,sha256=3KA2ZeYYdcDweJhUGopxu9EFgARhu2aWft5S5cj8Bc8,323
|
|
293
296
|
verifiers/v1/mcp/__init__.py,sha256=pm8ZWBiL1q1tY4wHvJzM_QrZ9oR8aGm-bIRWiRJsgHI,408
|
|
294
|
-
verifiers/v1/mcp/launch.py,sha256=
|
|
297
|
+
verifiers/v1/mcp/launch.py,sha256=KtRr59FMQkp4_m8ImUw-pIrxMUbDXd1RvzLJXw_HQ3Q,18452
|
|
295
298
|
verifiers/v1/mcp/server.py,sha256=zS3-m-0IX3z8nVzUhQj391f8_OieE1nKr5S4Nj0dSSU,10987
|
|
296
299
|
verifiers/v1/mcp/toolset.py,sha256=tSPXgTNREZyaeCnoFVnDfXxFAj_pStZMPUIgyPrY_Dk,996
|
|
297
300
|
verifiers/v1/runtimes/__init__.py,sha256=TqNtJVahzQuLS4fgCVeT9xdOzTkp0hRAPOkl7eFVDUg,2011
|
|
@@ -319,7 +322,7 @@ verifiers/v1/tasksets/textarena/__init__.py,sha256=Os2OlBY_pSH0B1DP32RitumgSLpq2
|
|
|
319
322
|
verifiers/v1/tasksets/textarena/taskset.py,sha256=9fa_unlFiFZyuN4I02rW3x4qyRnFWXqwTUZTloQefu0,4433
|
|
320
323
|
verifiers/v1/utils/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
321
324
|
verifiers/v1/utils/aio.py,sha256=0yNFOqB2oO6Qktc6NI_ZSAGuo1aHsIK_wuT1BtmoZ3I,1532
|
|
322
|
-
verifiers/v1/utils/compile.py,sha256=
|
|
325
|
+
verifiers/v1/utils/compile.py,sha256=Oh4hDVRNsql_BSOMh4kDEuze9_ZF17Q7uzBhhU70UZA,5663
|
|
323
326
|
verifiers/v1/utils/format.py,sha256=NfQo9M5KMzWCZpQaet5YtsJx56vCKAZPfbjZZJLJhhM,2187
|
|
324
327
|
verifiers/v1/utils/generic.py,sha256=2VIN9RapMO9pVkFHB1UKUHck1vlTJ1pjfad2NzxuuXA,2055
|
|
325
328
|
verifiers/v1/utils/git.py,sha256=MTRinOvG4EddfjdahHkFZUihHmhJ37n5ROtgjzjE1NA,8366
|
|
@@ -330,8 +333,8 @@ verifiers/v1/utils/logging.py,sha256=OcMHA6NsYux3oIzjPuI95rDWmFBHNcHDjZeNIhXTX-Y
|
|
|
330
333
|
verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
|
|
331
334
|
verifiers/v1/utils/sampling.py,sha256=52TJ5-Hmpvp4YSgKNR-LmAPVwm4RDcH-o2-BnbVTmYQ,965
|
|
332
335
|
verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
|
|
333
|
-
verifiers-0.2.2.
|
|
334
|
-
verifiers-0.2.2.
|
|
335
|
-
verifiers-0.2.2.
|
|
336
|
-
verifiers-0.2.2.
|
|
337
|
-
verifiers-0.2.2.
|
|
336
|
+
verifiers-0.2.2.dev57.dist-info/METADATA,sha256=4FbkVpDnb3GnGhrNC-YaPVwF8qfnOQgOMoHLmWkC57E,4545
|
|
337
|
+
verifiers-0.2.2.dev57.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
338
|
+
verifiers-0.2.2.dev57.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
|
|
339
|
+
verifiers-0.2.2.dev57.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
340
|
+
verifiers-0.2.2.dev57.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|