anybrowser 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- anybrowser/__init__.py +64 -0
- anybrowser/agent/__init__.py +32 -0
- anybrowser/agent/llm_planner.py +233 -0
- anybrowser/agent/memory/__init__.py +1 -0
- anybrowser/agent/planner.py +75 -0
- anybrowser/agent/runner.py +262 -0
- anybrowser/agent/tool.py +105 -0
- anybrowser/agent/tools/__init__.py +27 -0
- anybrowser/agent/tools/browser.py +234 -0
- anybrowser/cli/__init__.py +5 -0
- anybrowser/cli/main.py +256 -0
- anybrowser/core/__init__.py +62 -0
- anybrowser/core/engine.py +389 -0
- anybrowser/core/errors.py +136 -0
- anybrowser/core/registry.py +111 -0
- anybrowser/core/sync.py +119 -0
- anybrowser/core/types.py +241 -0
- anybrowser/daemon/__init__.py +29 -0
- anybrowser/daemon/client.py +459 -0
- anybrowser/daemon/codec.py +253 -0
- anybrowser/daemon/protocol.py +128 -0
- anybrowser/daemon/server.py +421 -0
- anybrowser/daemon/transport.py +130 -0
- anybrowser/daemon/transports/__init__.py +14 -0
- anybrowser/daemon/transports/base.py +156 -0
- anybrowser/daemon/transports/memory.py +128 -0
- anybrowser/daemon/transports/unix.py +196 -0
- anybrowser/daemon/transports/websocket.py +145 -0
- anybrowser/engines/__init__.py +1 -0
- anybrowser/engines/chrome/__init__.py +30 -0
- anybrowser/engines/chrome/cdp.py +459 -0
- anybrowser/engines/chrome/engine.py +1072 -0
- anybrowser/engines/playwright/__init__.py +5 -0
- anybrowser/engines/playwright/engine.py +601 -0
- anybrowser/engines/playwright/script.py +165 -0
- anybrowser/engines/safari/__init__.py +33 -0
- anybrowser/engines/safari/bridge.py +210 -0
- anybrowser/engines/safari/build.py +128 -0
- anybrowser/engines/safari/channel.py +153 -0
- anybrowser/engines/safari/engine.py +531 -0
- anybrowser/engines/safari/mac_engine/AnyBrowserSafariEngine.swift +698 -0
- anybrowser/engines/safari/mac_engine/Info.plist.in +48 -0
- anybrowser/models/__init__.py +16 -0
- anybrowser/models/anthropic.py +347 -0
- anybrowser/models/openai_compatible.py +347 -0
- anybrowser/models/provider.py +130 -0
- anybrowser/perception/__init__.py +20 -0
- anybrowser/perception/assets.py +45 -0
- anybrowser/perception/dom.py +90 -0
- anybrowser/perception/js/collect-elements.js +162 -0
- anybrowser/perception/js/deep-dom-helpers.js +594 -0
- anybrowser/perception/js/read.js +155 -0
- anybrowser-0.1.0.dist-info/METADATA +245 -0
- anybrowser-0.1.0.dist-info/RECORD +67 -0
- anybrowser-0.1.0.dist-info/WHEEL +4 -0
- anybrowser-0.1.0.dist-info/entry_points.txt +20 -0
- anybrowser-0.1.0.dist-info/licenses/LICENSE +202 -0
- anybrowser-0.1.0.dist-info/licenses/NOTICE +8 -0
- anybrowser_conformance/__init__.py +18 -0
- anybrowser_conformance/pages/drag.html +24 -0
- anybrowser_conformance/pages/frames.html +5 -0
- anybrowser_conformance/pages/index.html +39 -0
- anybrowser_conformance/pages/second.html +4 -0
- anybrowser_conformance/plugin.py +365 -0
- anybrowser_conformance/test_contract.py +341 -0
- anybrowser_conformance/test_model.py +120 -0
- anybrowser_conformance/test_transport.py +168 -0
anybrowser/__init__.py
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""AnyBrowser -- a pluggable browser-automation and agent runtime.
|
|
2
|
+
|
|
3
|
+
Three interfaces, each with an entry-point registry, each with a conformance
|
|
4
|
+
suite you can run against your own implementation:
|
|
5
|
+
|
|
6
|
+
* :class:`anybrowser.core.BrowserEngine` -- drive a browser (Chrome, Safari, yours)
|
|
7
|
+
* :class:`anybrowser.daemon.DaemonTransport` -- how clients reach the daemon
|
|
8
|
+
* :class:`anybrowser.models.ModelProvider` -- where completions come from
|
|
9
|
+
|
|
10
|
+
from anybrowser import open_engine
|
|
11
|
+
|
|
12
|
+
async with await open_engine("chrome") as engine:
|
|
13
|
+
await engine.navigate("https://example.com")
|
|
14
|
+
page = await engine.snapshot()
|
|
15
|
+
print(page.title, len(page.elements), "elements")
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from typing import Any
|
|
21
|
+
|
|
22
|
+
# NOT `from .core import engines`: the `anybrowser.engines` SUBPACKAGE binds
|
|
23
|
+
# that same name on this module as soon as anything imports it, silently
|
|
24
|
+
# replacing the registry with the package. In a source tree it survives on
|
|
25
|
+
# import-order luck; from an installed wheel it does not. Import the
|
|
26
|
+
# registry under a name nothing else claims.
|
|
27
|
+
from .core import BrowserEngine, Capability, SyncEngine
|
|
28
|
+
from .core.errors import AnyBrowserError
|
|
29
|
+
from .core.registry import engines as engine_registry
|
|
30
|
+
|
|
31
|
+
__version__ = "0.1.0"
|
|
32
|
+
|
|
33
|
+
__all__ = [
|
|
34
|
+
"AnyBrowserError",
|
|
35
|
+
"BrowserEngine",
|
|
36
|
+
"Capability",
|
|
37
|
+
"SyncEngine",
|
|
38
|
+
"__version__",
|
|
39
|
+
"available_engines",
|
|
40
|
+
"engines",
|
|
41
|
+
"open_engine",
|
|
42
|
+
]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
async def open_engine(name: str, /, **options: Any) -> BrowserEngine:
|
|
46
|
+
"""Construct and start the named engine.
|
|
47
|
+
|
|
48
|
+
``name`` is a key in the ``anybrowser.engines`` entry-point group. The returned engine
|
|
49
|
+
is already started and is also an async context manager, so both of these
|
|
50
|
+
work::
|
|
51
|
+
|
|
52
|
+
engine = await open_engine("chrome")
|
|
53
|
+
async with await open_engine("chrome") as engine: ...
|
|
54
|
+
"""
|
|
55
|
+
cls = engine_registry.get(name)
|
|
56
|
+
await cls.probe()
|
|
57
|
+
engine: BrowserEngine = cls(**options)
|
|
58
|
+
await engine.start()
|
|
59
|
+
return engine
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def available_engines() -> list[str]:
|
|
63
|
+
"""Every registered engine name, installed plugins included."""
|
|
64
|
+
return engine_registry.names()
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""The agent runtime: decide what to do, then do it.
|
|
2
|
+
|
|
3
|
+
runner = AgentRunner(engine, LLMPlanner(provider, model="claude-sonnet-5"))
|
|
4
|
+
result = await runner.run("find the pricing page")
|
|
5
|
+
|
|
6
|
+
Swapping the :class:`Planner` changes what kind of agent this is; the tools,
|
|
7
|
+
engines and daemon underneath do not move.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from .llm_planner import SYSTEM_PROMPT, LLMPlanner
|
|
11
|
+
from .planner import Decision, Observation, Planner, Step
|
|
12
|
+
from .runner import AgentRunner, RunConfig, RunResult, StopReason, render_history
|
|
13
|
+
from .tool import Tool, ToolContext, ToolResult
|
|
14
|
+
from .tools import default_tools
|
|
15
|
+
|
|
16
|
+
__all__ = [
|
|
17
|
+
"SYSTEM_PROMPT",
|
|
18
|
+
"AgentRunner",
|
|
19
|
+
"Decision",
|
|
20
|
+
"LLMPlanner",
|
|
21
|
+
"Observation",
|
|
22
|
+
"Planner",
|
|
23
|
+
"RunConfig",
|
|
24
|
+
"RunResult",
|
|
25
|
+
"Step",
|
|
26
|
+
"StopReason",
|
|
27
|
+
"Tool",
|
|
28
|
+
"ToolContext",
|
|
29
|
+
"ToolResult",
|
|
30
|
+
"default_tools",
|
|
31
|
+
"render_history",
|
|
32
|
+
]
|
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
"""A planner that asks a model what to do next.
|
|
2
|
+
|
|
3
|
+
The default planner. It renders the goal, the page, the history and the tool
|
|
4
|
+
schemas into one prompt, and parses a single JSON decision back.
|
|
5
|
+
|
|
6
|
+
Two choices here are deliberate and worth defending.
|
|
7
|
+
|
|
8
|
+
**JSON in the response, not provider tool-calling.** Tool-call formats differ
|
|
9
|
+
per provider in ways no wrapper survives intact, and several models AnyBrowser
|
|
10
|
+
should support have none at all. Asking for one JSON object works everywhere and
|
|
11
|
+
keeps :class:`~anybrowser.models.provider.ModelProvider` small.
|
|
12
|
+
|
|
13
|
+
**`narrative` is required.** A model that emits no reasoning tokens still has to
|
|
14
|
+
say why. A history reading "clicked, clicked, clicked" cannot be debugged by
|
|
15
|
+
anyone -- including the model itself on the next step.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import json
|
|
21
|
+
import re
|
|
22
|
+
from collections.abc import Sequence
|
|
23
|
+
from typing import Any
|
|
24
|
+
|
|
25
|
+
from ..core.errors import ModelError
|
|
26
|
+
from ..models.provider import ImagePart, Message, ModelProvider, Role
|
|
27
|
+
from .planner import Decision, Observation, Planner
|
|
28
|
+
from .runner import render_history
|
|
29
|
+
from .tool import Tool
|
|
30
|
+
|
|
31
|
+
__all__ = ["SYSTEM_PROMPT", "LLMPlanner"]
|
|
32
|
+
|
|
33
|
+
SYSTEM_PROMPT = """\
|
|
34
|
+
You are driving a web browser to accomplish a goal. You see the page as a list \
|
|
35
|
+
of interactive elements, each with a handle, plus a screenshot when one is \
|
|
36
|
+
available.
|
|
37
|
+
|
|
38
|
+
Choose exactly ONE action per turn. Reply with one JSON object and nothing else:
|
|
39
|
+
|
|
40
|
+
{"tool": "<tool name>", "arguments": {...}, "narrative": "<why, in one sentence>"}
|
|
41
|
+
|
|
42
|
+
For example, to click the element listed as `handle=3:12`:
|
|
43
|
+
|
|
44
|
+
{"tool": "click", "arguments": {"handle": "3:12"}, "narrative": "this is the login button"}
|
|
45
|
+
|
|
46
|
+
When the goal is achieved, reply instead with:
|
|
47
|
+
|
|
48
|
+
{"done": true, "answer": "<what you found, or what you did>", "narrative": "<why you are finished>"}
|
|
49
|
+
|
|
50
|
+
Rules that matter:
|
|
51
|
+
|
|
52
|
+
- Every element below is listed as `handle=<id>`. Pass that id EXACTLY, copied \
|
|
53
|
+
character for character. It is an opaque id such as "3:12" -- never the label, \
|
|
54
|
+
the tag, or anything you read off the screenshot.
|
|
55
|
+
- Act only on handles in the CURRENT listing. Handles from earlier turns are \
|
|
56
|
+
stale and will be refused.
|
|
57
|
+
- A result marked [no effect] means the action changed nothing. Doing it again \
|
|
58
|
+
will change nothing again. Try something else: scroll, look, or pick a \
|
|
59
|
+
different element.
|
|
60
|
+
- A result marked [failed] means the action could not run at all. Read the \
|
|
61
|
+
reason before choosing.
|
|
62
|
+
- "narrative" is required, always, and must say why you chose this action -- \
|
|
63
|
+
not what the action is.
|
|
64
|
+
- Do not claim the goal is done without evidence in the page that it is.\
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class LLMPlanner(Planner):
|
|
69
|
+
"""Ask a model for the next action."""
|
|
70
|
+
|
|
71
|
+
name = "llm"
|
|
72
|
+
|
|
73
|
+
def __init__(
|
|
74
|
+
self,
|
|
75
|
+
provider: ModelProvider,
|
|
76
|
+
*,
|
|
77
|
+
model: str,
|
|
78
|
+
max_elements: int = 60,
|
|
79
|
+
temperature: float = 0.0,
|
|
80
|
+
max_tokens: int = 800,
|
|
81
|
+
system_prompt: str = SYSTEM_PROMPT,
|
|
82
|
+
) -> None:
|
|
83
|
+
self._provider = provider
|
|
84
|
+
self._model = model
|
|
85
|
+
self._max_elements = max_elements
|
|
86
|
+
self._temperature = temperature
|
|
87
|
+
self._max_tokens = max_tokens
|
|
88
|
+
self._system_prompt = system_prompt
|
|
89
|
+
#: Accumulated across a run, so a caller can meter what an agent cost.
|
|
90
|
+
self.total_cost_usd = 0.0
|
|
91
|
+
self.total_tokens = 0
|
|
92
|
+
|
|
93
|
+
# ------------------------------------------------------------------ #
|
|
94
|
+
# Prompting #
|
|
95
|
+
# ------------------------------------------------------------------ #
|
|
96
|
+
|
|
97
|
+
def _render_page(self, observation: Observation) -> str:
|
|
98
|
+
page = observation.snapshot
|
|
99
|
+
lines = [f"URL: {page.url}", f"Title: {page.title}", "", "Elements:"]
|
|
100
|
+
if not page.elements:
|
|
101
|
+
lines.append(" (none found — try scrolling, or the page may still be loading)")
|
|
102
|
+
for element in page.elements[: self._max_elements]:
|
|
103
|
+
# `description` already starts with the tag or role, so printing
|
|
104
|
+
# both made the line read as `[2:0] a 'a Learn more'` -- and a model
|
|
105
|
+
# will reach for the quoted text as the identifier. Give the handle
|
|
106
|
+
# its own labelled column and print the label alone.
|
|
107
|
+
kind = element.role or element.tag
|
|
108
|
+
label = element.label or element.placeholder or element.value or "(no label)"
|
|
109
|
+
detail = f"handle={element.handle} <{kind}> {label!r}"
|
|
110
|
+
if element.value and element.value != label:
|
|
111
|
+
detail += f" value={element.value!r}"
|
|
112
|
+
if element.disabled:
|
|
113
|
+
detail += " (disabled)"
|
|
114
|
+
lines.append(" " + detail)
|
|
115
|
+
if len(page.elements) > self._max_elements:
|
|
116
|
+
hidden = len(page.elements) - self._max_elements
|
|
117
|
+
lines.append(f" … and {hidden} more not shown")
|
|
118
|
+
return "\n".join(lines)
|
|
119
|
+
|
|
120
|
+
def _render_tools(self, tools: Sequence[Tool]) -> str:
|
|
121
|
+
return "\n".join(
|
|
122
|
+
f"- {tool.name}: {tool.description}\n arguments: "
|
|
123
|
+
f"{json.dumps(tool.parameters.get('properties', {}))}"
|
|
124
|
+
for tool in tools
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
def _build_messages(self, observation: Observation, tools: Sequence[Tool]) -> list[Message]:
|
|
128
|
+
parts = [
|
|
129
|
+
f"GOAL: {observation.goal}",
|
|
130
|
+
"",
|
|
131
|
+
"TOOLS:",
|
|
132
|
+
self._render_tools(tools),
|
|
133
|
+
"",
|
|
134
|
+
"PAGE:",
|
|
135
|
+
self._render_page(observation),
|
|
136
|
+
]
|
|
137
|
+
if observation.history:
|
|
138
|
+
parts += ["", "WHAT YOU HAVE DONE:", render_history(observation.history)]
|
|
139
|
+
if observation.notes:
|
|
140
|
+
parts += ["", "NOTES:", *(f"- {n}" for n in observation.notes)]
|
|
141
|
+
parts += ["", f"Step {observation.step_index + 1}. Choose one action."]
|
|
142
|
+
|
|
143
|
+
images: list[ImagePart] = []
|
|
144
|
+
if observation.screenshot is not None and self._provider.supports_images(self._model):
|
|
145
|
+
images.append(
|
|
146
|
+
ImagePart(
|
|
147
|
+
data=observation.screenshot.data,
|
|
148
|
+
media_type=f"image/{observation.screenshot.format}",
|
|
149
|
+
)
|
|
150
|
+
)
|
|
151
|
+
return [
|
|
152
|
+
Message(role=Role.SYSTEM, text=self._system_prompt),
|
|
153
|
+
Message(role=Role.USER, text="\n".join(parts), images=images),
|
|
154
|
+
]
|
|
155
|
+
|
|
156
|
+
# ------------------------------------------------------------------ #
|
|
157
|
+
# Deciding #
|
|
158
|
+
# ------------------------------------------------------------------ #
|
|
159
|
+
|
|
160
|
+
async def decide(self, observation: Observation, tools: Sequence[Tool]) -> Decision:
|
|
161
|
+
messages = self._build_messages(observation, tools)
|
|
162
|
+
completion = await self._provider.complete(
|
|
163
|
+
messages,
|
|
164
|
+
model=self._model,
|
|
165
|
+
temperature=self._temperature,
|
|
166
|
+
max_tokens=self._max_tokens,
|
|
167
|
+
)
|
|
168
|
+
self.total_cost_usd += completion.usage.cost_usd
|
|
169
|
+
self.total_tokens += completion.usage.input_tokens + completion.usage.output_tokens
|
|
170
|
+
return self._parse(completion.text, tools)
|
|
171
|
+
|
|
172
|
+
def _parse(self, text: str, tools: Sequence[Tool]) -> Decision:
|
|
173
|
+
payload = _extract_json(text)
|
|
174
|
+
if payload is None:
|
|
175
|
+
raise ModelError(f"planner did not return JSON: {text[:300]!r}")
|
|
176
|
+
|
|
177
|
+
narrative = str(payload.get("narrative") or "").strip()
|
|
178
|
+
if payload.get("done"):
|
|
179
|
+
return Decision(
|
|
180
|
+
tool="",
|
|
181
|
+
arguments={},
|
|
182
|
+
# A model that finishes without saying why still has to say
|
|
183
|
+
# something; refusing here would fail a successful run over
|
|
184
|
+
# a missing sentence.
|
|
185
|
+
narrative=narrative or "the goal appears to be met",
|
|
186
|
+
done=True,
|
|
187
|
+
answer=str(payload.get("answer") or ""),
|
|
188
|
+
)
|
|
189
|
+
|
|
190
|
+
tool = str(payload.get("tool") or "").strip()
|
|
191
|
+
if not tool:
|
|
192
|
+
raise ModelError(f"planner named no tool: {text[:300]!r}")
|
|
193
|
+
known = {t.name for t in tools}
|
|
194
|
+
if tool not in known:
|
|
195
|
+
raise ModelError(f"planner chose unknown tool {tool!r}; available: {sorted(known)}")
|
|
196
|
+
|
|
197
|
+
arguments = payload.get("arguments")
|
|
198
|
+
if not isinstance(arguments, dict):
|
|
199
|
+
arguments = {}
|
|
200
|
+
return Decision(
|
|
201
|
+
tool=tool,
|
|
202
|
+
arguments=arguments,
|
|
203
|
+
narrative=narrative or f"calling {tool}",
|
|
204
|
+
)
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
_FENCE = re.compile(r"```(?:json)?\s*(.*?)```", re.S)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def _extract_json(text: str) -> dict[str, Any] | None:
|
|
211
|
+
"""Find the JSON object in a model's reply.
|
|
212
|
+
|
|
213
|
+
Models wrap JSON in prose and fences no matter how firmly the prompt asks
|
|
214
|
+
them not to. Being lenient here costs nothing; being strict costs a run.
|
|
215
|
+
"""
|
|
216
|
+
candidates: list[str] = []
|
|
217
|
+
fenced = _FENCE.search(text)
|
|
218
|
+
if fenced:
|
|
219
|
+
candidates.append(fenced.group(1))
|
|
220
|
+
candidates.append(text)
|
|
221
|
+
start = text.find("{")
|
|
222
|
+
end = text.rfind("}")
|
|
223
|
+
if start != -1 and end > start:
|
|
224
|
+
candidates.append(text[start : end + 1])
|
|
225
|
+
|
|
226
|
+
for candidate in candidates:
|
|
227
|
+
try:
|
|
228
|
+
parsed = json.loads(candidate.strip())
|
|
229
|
+
except (json.JSONDecodeError, ValueError):
|
|
230
|
+
continue
|
|
231
|
+
if isinstance(parsed, dict):
|
|
232
|
+
return parsed
|
|
233
|
+
return None
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""AnyBrowser subpackage."""
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
"""``Planner`` -- decides the next action.
|
|
2
|
+
|
|
3
|
+
Swapping the planner is how you change what kind of agent this is. The default
|
|
4
|
+
one is an LLM reading a screenshot and a snapshot; a scripted planner replaying a
|
|
5
|
+
recorded workflow, or a heuristic one, implements the same two methods and runs
|
|
6
|
+
on the same executor, tools and engines.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import abc
|
|
12
|
+
from collections.abc import Mapping, Sequence
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
from typing import Any
|
|
15
|
+
|
|
16
|
+
from ..core.types import Screenshot, Snapshot
|
|
17
|
+
from .tool import Tool, ToolResult
|
|
18
|
+
|
|
19
|
+
__all__ = ["Decision", "Observation", "Planner"]
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(frozen=True, slots=True)
|
|
23
|
+
class Observation:
|
|
24
|
+
"""Everything the planner is allowed to see this step."""
|
|
25
|
+
|
|
26
|
+
goal: str
|
|
27
|
+
snapshot: Snapshot
|
|
28
|
+
screenshot: Screenshot | None = None
|
|
29
|
+
history: Sequence[Step] = ()
|
|
30
|
+
notes: Sequence[str] = ()
|
|
31
|
+
step_index: int = 0
|
|
32
|
+
extra: Mapping[str, Any] = field(default_factory=dict)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass(frozen=True, slots=True)
|
|
36
|
+
class Decision:
|
|
37
|
+
"""One chosen action.
|
|
38
|
+
|
|
39
|
+
``narrative`` is required and must be a human explanation of *why*, not a
|
|
40
|
+
restatement of the tool call. Models with no visible reasoning still have to
|
|
41
|
+
produce one; a run whose history reads "clicked, clicked, clicked" cannot be
|
|
42
|
+
debugged by anyone, including the model itself on the next step.
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
tool: str
|
|
46
|
+
arguments: Mapping[str, Any]
|
|
47
|
+
narrative: str
|
|
48
|
+
done: bool = False
|
|
49
|
+
answer: str = ""
|
|
50
|
+
|
|
51
|
+
def __post_init__(self) -> None:
|
|
52
|
+
if not self.narrative.strip():
|
|
53
|
+
raise ValueError("Decision.narrative is required")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@dataclass(frozen=True, slots=True)
|
|
57
|
+
class Step:
|
|
58
|
+
"""One completed (decision, result) pair, as history renders it."""
|
|
59
|
+
|
|
60
|
+
decision: Decision
|
|
61
|
+
result: ToolResult
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class Planner(abc.ABC):
|
|
65
|
+
"""Chooses the next tool call."""
|
|
66
|
+
|
|
67
|
+
#: Registry name, if published as a plugin.
|
|
68
|
+
name: str = ""
|
|
69
|
+
|
|
70
|
+
@abc.abstractmethod
|
|
71
|
+
async def decide(self, observation: Observation, tools: Sequence[Tool]) -> Decision: ...
|
|
72
|
+
|
|
73
|
+
async def reflect(self, observation: Observation, step: Step) -> None:
|
|
74
|
+
"""Optional hook after each step, for memory or a stuck detector."""
|
|
75
|
+
return None
|
|
@@ -0,0 +1,262 @@
|
|
|
1
|
+
"""The loop: observe, decide, act, remember.
|
|
2
|
+
|
|
3
|
+
runner = AgentRunner(engine, planner)
|
|
4
|
+
result = await runner.run("find the pricing page and read the top tier")
|
|
5
|
+
|
|
6
|
+
Deliberately small. Everything that decides *quality* lives in the planner and
|
|
7
|
+
in perception; this file's whole job is to be an honest, interruptible loop that
|
|
8
|
+
never lies to the planner about what happened.
|
|
9
|
+
|
|
10
|
+
Three properties it guarantees, each of which is a failure mode somewhere else:
|
|
11
|
+
|
|
12
|
+
**It re-observes before every decision.** A planner reasoning about a snapshot
|
|
13
|
+
taken three actions ago is reasoning about a page that no longer exists, and
|
|
14
|
+
its handles are stale by definition.
|
|
15
|
+
|
|
16
|
+
**It records no-ops as no-ops.** ``ToolResult.changed`` goes into history
|
|
17
|
+
verbatim. A history that renders "clicked, success" for a click that hit an
|
|
18
|
+
invisible overlay teaches the model that clicking again is progress, and it
|
|
19
|
+
will do that until the budget runs out.
|
|
20
|
+
|
|
21
|
+
**It stops.** A budget, a stop signal, and a repetition breaker. Long-horizon
|
|
22
|
+
agents do not fail by crashing; they fail by continuing.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import asyncio
|
|
28
|
+
import logging
|
|
29
|
+
import time
|
|
30
|
+
from collections.abc import Callable, Sequence
|
|
31
|
+
from dataclasses import dataclass, field
|
|
32
|
+
|
|
33
|
+
from ..core.engine import BrowserEngine
|
|
34
|
+
from ..core.errors import AnyBrowserError
|
|
35
|
+
from ..core.types import Screenshot
|
|
36
|
+
from .planner import Decision, Observation, Planner, Step
|
|
37
|
+
from .tool import Tool, ToolContext, ToolResult
|
|
38
|
+
from .tools import default_tools
|
|
39
|
+
|
|
40
|
+
logger = logging.getLogger(__name__)
|
|
41
|
+
|
|
42
|
+
__all__ = ["AgentRunner", "RunConfig", "RunResult", "StopReason"]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class StopReason(str):
|
|
46
|
+
"""Why a run ended. A plain string subclass so it renders itself in logs."""
|
|
47
|
+
|
|
48
|
+
DONE = "done"
|
|
49
|
+
BUDGET = "budget_exhausted"
|
|
50
|
+
STOPPED = "stopped"
|
|
51
|
+
STUCK = "stuck"
|
|
52
|
+
ERROR = "error"
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@dataclass(slots=True)
|
|
56
|
+
class RunConfig:
|
|
57
|
+
"""Limits. All of them exist because something ran away without them."""
|
|
58
|
+
|
|
59
|
+
#: Hard ceiling on actions. The single most important safety valve.
|
|
60
|
+
max_steps: int = 40
|
|
61
|
+
#: Wall-clock ceiling, in seconds. Zero disables it.
|
|
62
|
+
max_seconds: float = 600.0
|
|
63
|
+
#: Give the planner a screenshot alongside the snapshot.
|
|
64
|
+
include_screenshot: bool = True
|
|
65
|
+
#: How many consecutive actions may change nothing before the run is called
|
|
66
|
+
#: stuck. Counts any action, not just repeats -- see AgentRunner.run for why
|
|
67
|
+
#: repetition is the wrong thing to measure. Four leaves room for a genuine
|
|
68
|
+
#: retry and a look in between.
|
|
69
|
+
repeat_limit: int = 4
|
|
70
|
+
#: Confirm before a mutating tool runs. Returning False refuses the action;
|
|
71
|
+
#: the planner is told, and can choose something else.
|
|
72
|
+
confirm: Callable[[Decision], bool] | None = None
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@dataclass(slots=True)
|
|
76
|
+
class RunResult:
|
|
77
|
+
goal: str
|
|
78
|
+
answer: str = ""
|
|
79
|
+
stop_reason: str = StopReason.DONE
|
|
80
|
+
steps: list[Step] = field(default_factory=list)
|
|
81
|
+
error: str = ""
|
|
82
|
+
elapsed: float = 0.0
|
|
83
|
+
|
|
84
|
+
@property
|
|
85
|
+
def ok(self) -> bool:
|
|
86
|
+
return self.stop_reason == StopReason.DONE
|
|
87
|
+
|
|
88
|
+
@property
|
|
89
|
+
def step_count(self) -> int:
|
|
90
|
+
return len(self.steps)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
class AgentRunner:
|
|
94
|
+
"""Drives a planner against a browser until it is done or out of budget."""
|
|
95
|
+
|
|
96
|
+
def __init__(
|
|
97
|
+
self,
|
|
98
|
+
engine: BrowserEngine,
|
|
99
|
+
planner: Planner,
|
|
100
|
+
*,
|
|
101
|
+
tools: Sequence[Tool] | None = None,
|
|
102
|
+
config: RunConfig | None = None,
|
|
103
|
+
) -> None:
|
|
104
|
+
self._engine = engine
|
|
105
|
+
self._planner = planner
|
|
106
|
+
self._tools = list(tools) if tools is not None else default_tools()
|
|
107
|
+
self._config = config or RunConfig()
|
|
108
|
+
self._by_name = {tool.name: tool for tool in self._tools}
|
|
109
|
+
self._stop = asyncio.Event()
|
|
110
|
+
|
|
111
|
+
@property
|
|
112
|
+
def tools(self) -> Sequence[Tool]:
|
|
113
|
+
return tuple(self._tools)
|
|
114
|
+
|
|
115
|
+
def stop(self) -> None:
|
|
116
|
+
"""Ask the run to finish after the current action.
|
|
117
|
+
|
|
118
|
+
Cooperative rather than a cancellation, because killing an agent
|
|
119
|
+
mid-action leaves the page in a state nobody recorded.
|
|
120
|
+
"""
|
|
121
|
+
self._stop.set()
|
|
122
|
+
|
|
123
|
+
async def run(self, goal: str, *, notes: Sequence[str] = ()) -> RunResult:
|
|
124
|
+
started = time.monotonic()
|
|
125
|
+
result = RunResult(goal=goal)
|
|
126
|
+
context = ToolContext(engine=self._engine)
|
|
127
|
+
recent: list[tuple[str, str]] = []
|
|
128
|
+
|
|
129
|
+
for index in range(self._config.max_steps):
|
|
130
|
+
if self._stop.is_set():
|
|
131
|
+
result.stop_reason = StopReason.STOPPED
|
|
132
|
+
break
|
|
133
|
+
if self._config.max_seconds and time.monotonic() - started > self._config.max_seconds:
|
|
134
|
+
result.stop_reason = StopReason.BUDGET
|
|
135
|
+
break
|
|
136
|
+
|
|
137
|
+
try:
|
|
138
|
+
observation = await self._observe(goal, result.steps, notes, index)
|
|
139
|
+
except AnyBrowserError as exc:
|
|
140
|
+
result.stop_reason, result.error = StopReason.ERROR, str(exc)
|
|
141
|
+
break
|
|
142
|
+
context.snapshot = observation.snapshot
|
|
143
|
+
|
|
144
|
+
try:
|
|
145
|
+
decision = await self._planner.decide(observation, self._tools)
|
|
146
|
+
except AnyBrowserError as exc:
|
|
147
|
+
result.stop_reason, result.error = StopReason.ERROR, str(exc)
|
|
148
|
+
break
|
|
149
|
+
|
|
150
|
+
if decision.done:
|
|
151
|
+
result.answer = decision.answer
|
|
152
|
+
result.stop_reason = StopReason.DONE
|
|
153
|
+
result.steps.append(
|
|
154
|
+
Step(
|
|
155
|
+
decision=decision, result=ToolResult(ok=True, changed=False, summary="done")
|
|
156
|
+
)
|
|
157
|
+
)
|
|
158
|
+
break
|
|
159
|
+
|
|
160
|
+
step_result = await self._act(context, decision)
|
|
161
|
+
step = Step(decision=decision, result=step_result)
|
|
162
|
+
result.steps.append(step)
|
|
163
|
+
await self._planner.reflect(observation, step)
|
|
164
|
+
|
|
165
|
+
# Stuck detection, keyed on *progress* rather than on repetition.
|
|
166
|
+
#
|
|
167
|
+
# An earlier version broke only on identical consecutive actions,
|
|
168
|
+
# and a live run walked straight past it: the commonest loop is
|
|
169
|
+
# act, look, act, look -- never two the same in a row, and never
|
|
170
|
+
# getting anywhere. Any step that changes something clears the
|
|
171
|
+
# window, so what this measures is simply "nothing has happened for
|
|
172
|
+
# N actions", whatever shape they took.
|
|
173
|
+
signature = (decision.tool, repr(sorted(decision.arguments.items())))
|
|
174
|
+
if step_result.changed:
|
|
175
|
+
recent.clear()
|
|
176
|
+
else:
|
|
177
|
+
recent.append(signature)
|
|
178
|
+
if len(recent) >= self._config.repeat_limit:
|
|
179
|
+
tools_tried = ", ".join(sorted({name for name, _ in recent}))
|
|
180
|
+
result.stop_reason = StopReason.STUCK
|
|
181
|
+
result.error = (
|
|
182
|
+
f"{len(recent)} actions with no effect on the page (tried: {tools_tried})"
|
|
183
|
+
)
|
|
184
|
+
break
|
|
185
|
+
else:
|
|
186
|
+
result.stop_reason = StopReason.BUDGET
|
|
187
|
+
|
|
188
|
+
result.elapsed = time.monotonic() - started
|
|
189
|
+
# Cleared at the END, not the start: a stop requested before this run
|
|
190
|
+
# began -- or between runs -- must be honoured, not discarded by the
|
|
191
|
+
# very call it was meant to stop.
|
|
192
|
+
self._stop.clear()
|
|
193
|
+
return result
|
|
194
|
+
|
|
195
|
+
# ------------------------------------------------------------------ #
|
|
196
|
+
# Internals #
|
|
197
|
+
# ------------------------------------------------------------------ #
|
|
198
|
+
|
|
199
|
+
async def _observe(
|
|
200
|
+
self, goal: str, history: Sequence[Step], notes: Sequence[str], index: int
|
|
201
|
+
) -> Observation:
|
|
202
|
+
snapshot = await self._engine.snapshot()
|
|
203
|
+
screenshot: Screenshot | None = None
|
|
204
|
+
if self._config.include_screenshot:
|
|
205
|
+
try:
|
|
206
|
+
screenshot = await self._engine.screenshot()
|
|
207
|
+
except AnyBrowserError:
|
|
208
|
+
# A planner that can read a snapshot can work without pixels;
|
|
209
|
+
# losing the screenshot should degrade the run, not end it.
|
|
210
|
+
logger.debug("screenshot unavailable", exc_info=True)
|
|
211
|
+
return Observation(
|
|
212
|
+
goal=goal,
|
|
213
|
+
snapshot=snapshot,
|
|
214
|
+
screenshot=screenshot,
|
|
215
|
+
history=tuple(history),
|
|
216
|
+
notes=tuple(notes),
|
|
217
|
+
step_index=index,
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
async def _act(self, context: ToolContext, decision: Decision) -> ToolResult:
|
|
221
|
+
tool = self._by_name.get(decision.tool)
|
|
222
|
+
if tool is None:
|
|
223
|
+
known = ", ".join(sorted(self._by_name))
|
|
224
|
+
return ToolResult.failure(f"no tool named {decision.tool!r}; available: {known}")
|
|
225
|
+
|
|
226
|
+
# The harness raises the confirmation, not the agent. A prompt-level
|
|
227
|
+
# "ask before destructive actions" rule is advisory and leaks; a gate at
|
|
228
|
+
# dispatch cannot be talked out of.
|
|
229
|
+
gated = tool.mutating and self._config.confirm is not None
|
|
230
|
+
if gated and not self._config.confirm(decision): # type: ignore[misc]
|
|
231
|
+
return ToolResult.failure("refused: the action was not confirmed")
|
|
232
|
+
|
|
233
|
+
try:
|
|
234
|
+
return await tool.run(context, **dict(decision.arguments))
|
|
235
|
+
except TypeError as exc:
|
|
236
|
+
return ToolResult.failure(f"bad arguments for {decision.tool}: {exc}")
|
|
237
|
+
except AnyBrowserError as exc:
|
|
238
|
+
return ToolResult.failure(str(exc), error_type=type(exc).__name__)
|
|
239
|
+
except Exception as exc: # a third-party tool may raise anything
|
|
240
|
+
logger.exception("tool %s failed", decision.tool)
|
|
241
|
+
return ToolResult.failure(f"{decision.tool} raised: {exc}")
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def render_history(steps: Sequence[Step], *, limit: int = 20) -> str:
|
|
245
|
+
"""Format history for a prompt.
|
|
246
|
+
|
|
247
|
+
The ``[no effect]`` marker is the whole point of this function. Without it
|
|
248
|
+
every line reads as progress and the model has no way to tell a working
|
|
249
|
+
action from one that has been failing silently for five turns.
|
|
250
|
+
"""
|
|
251
|
+
lines: list[str] = []
|
|
252
|
+
for index, step in enumerate(steps[-limit:], start=max(1, len(steps) - limit + 1)):
|
|
253
|
+
marker = "" if step.result.changed else " [no effect]"
|
|
254
|
+
if not step.result.ok:
|
|
255
|
+
marker = " [failed]"
|
|
256
|
+
arguments = ", ".join(f"{k}={v!r}" for k, v in step.decision.arguments.items())
|
|
257
|
+
lines.append(
|
|
258
|
+
f"{index}. {step.decision.tool}({arguments}){marker}\n"
|
|
259
|
+
f" why: {step.decision.narrative}\n"
|
|
260
|
+
f" result: {step.result.summary}"
|
|
261
|
+
)
|
|
262
|
+
return "\n".join(lines)
|