vco 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- vco/__init__.py +35 -0
- vco/adaptive_zoom.py +245 -0
- vco/backends/apple_vision_ocr.swift +73 -0
- vco/browser.py +341 -0
- vco/capture.py +31 -0
- vco/cli.py +1187 -0
- vco/executor.py +81 -0
- vco/geometry.py +106 -0
- vco/grid.py +108 -0
- vco/loop.py +132 -0
- vco/mcp_server.py +618 -0
- vco/models.py +181 -0
- vco/ocr.py +417 -0
- vco/ocr_assist.py +193 -0
- vco/ocr_server.py +111 -0
- vco/providers.py +1411 -0
- vco/webagent.py +420 -0
- vco/webdebug.py +180 -0
- vco/zoom.py +189 -0
- vco-0.1.0.dist-info/METADATA +387 -0
- vco-0.1.0.dist-info/RECORD +24 -0
- vco-0.1.0.dist-info/WHEEL +5 -0
- vco-0.1.0.dist-info/entry_points.txt +2 -0
- vco-0.1.0.dist-info/top_level.txt +1 -0
vco/__init__.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""vco — a clicker for LLMs: operate web pages and desktop screens."""
|
|
2
|
+
|
|
3
|
+
from .geometry import GridMapper
|
|
4
|
+
from .models import Action, GridPoint, GridSpec, Region, parse_action
|
|
5
|
+
from .ocr import (
|
|
6
|
+
HTTPOCRBackend,
|
|
7
|
+
OCRBackend,
|
|
8
|
+
OCRBox,
|
|
9
|
+
OCRRequest,
|
|
10
|
+
OCRResult,
|
|
11
|
+
RapidOCRBackend,
|
|
12
|
+
create_ocr_backend,
|
|
13
|
+
register_ocr_backend,
|
|
14
|
+
)
|
|
15
|
+
from .ocr_assist import OCRAssistProvider
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"Action",
|
|
19
|
+
"GridMapper",
|
|
20
|
+
"GridPoint",
|
|
21
|
+
"GridSpec",
|
|
22
|
+
"OCRBackend",
|
|
23
|
+
"OCRAssistProvider",
|
|
24
|
+
"OCRBox",
|
|
25
|
+
"OCRRequest",
|
|
26
|
+
"OCRResult",
|
|
27
|
+
"HTTPOCRBackend",
|
|
28
|
+
"RapidOCRBackend",
|
|
29
|
+
"Region",
|
|
30
|
+
"create_ocr_backend",
|
|
31
|
+
"parse_action",
|
|
32
|
+
"register_ocr_backend",
|
|
33
|
+
]
|
|
34
|
+
|
|
35
|
+
__version__ = "0.1.0"
|
vco/adaptive_zoom.py
ADDED
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
"""Model-directed recursive zoom for dense-grid visual localization."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import math
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from typing import Tuple
|
|
8
|
+
|
|
9
|
+
from PIL import Image
|
|
10
|
+
|
|
11
|
+
from .geometry import GridMapper
|
|
12
|
+
from .grid import render_numbered_grid
|
|
13
|
+
from .models import Action, GridSpec, Region, parse_action
|
|
14
|
+
from .zoom import neighborhood_crop_box
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
Box = Tuple[int, int, int, int]
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class AdaptiveZoomLevel:
|
|
22
|
+
level: int
|
|
23
|
+
view_box: Box
|
|
24
|
+
decision: dict
|
|
25
|
+
next_view_box: Box | None
|
|
26
|
+
provider_metadata: dict | None
|
|
27
|
+
|
|
28
|
+
def as_dict(self) -> dict:
|
|
29
|
+
return {
|
|
30
|
+
"level": self.level,
|
|
31
|
+
"view_box_image_pixels": self.view_box,
|
|
32
|
+
"decision": self.decision,
|
|
33
|
+
"next_view_box_image_pixels": self.next_view_box,
|
|
34
|
+
"provider_metadata": self.provider_metadata,
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass(frozen=True)
|
|
39
|
+
class AdaptiveZoomTrace:
|
|
40
|
+
levels: tuple[AdaptiveZoomLevel, ...]
|
|
41
|
+
final_action: dict
|
|
42
|
+
|
|
43
|
+
def as_dict(self) -> dict:
|
|
44
|
+
return {
|
|
45
|
+
"mode": "model-directed-adaptive-zoom",
|
|
46
|
+
"levels": [level.as_dict() for level in self.levels],
|
|
47
|
+
"final_action": self.final_action,
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _map_crop_to_original(display_crop: Box, view_box: Box, display_size) -> Box:
|
|
52
|
+
display_width, display_height = display_size
|
|
53
|
+
view_left, view_top, view_right, view_bottom = view_box
|
|
54
|
+
view_width = view_right - view_left
|
|
55
|
+
view_height = view_bottom - view_top
|
|
56
|
+
left, top, right, bottom = display_crop
|
|
57
|
+
mapped = (
|
|
58
|
+
view_left + math.floor(left * view_width / display_width),
|
|
59
|
+
view_top + math.floor(top * view_height / display_height),
|
|
60
|
+
view_left + math.ceil(right * view_width / display_width),
|
|
61
|
+
view_top + math.ceil(bottom * view_height / display_height),
|
|
62
|
+
)
|
|
63
|
+
if mapped[2] <= mapped[0] or mapped[3] <= mapped[1]:
|
|
64
|
+
raise ValueError("adaptive zoom produced an empty crop")
|
|
65
|
+
return mapped
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _map_point_to_original(point, view_box: Box, display_size):
|
|
69
|
+
x, y = point
|
|
70
|
+
display_width, display_height = display_size
|
|
71
|
+
left, top, right, bottom = view_box
|
|
72
|
+
view_width = right - left
|
|
73
|
+
view_height = bottom - top
|
|
74
|
+
original_x = left + round(x * max(0, view_width - 1) / max(1, display_width - 1))
|
|
75
|
+
original_y = top + round(y * max(0, view_height - 1) / max(1, display_height - 1))
|
|
76
|
+
return original_x, original_y
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class AdaptiveZoomProvider:
|
|
80
|
+
"""Ask the model to click now or request another zoom level."""
|
|
81
|
+
|
|
82
|
+
def __init__(
|
|
83
|
+
self,
|
|
84
|
+
provider,
|
|
85
|
+
*,
|
|
86
|
+
max_levels: int = 3,
|
|
87
|
+
span_cells: float = 4.0,
|
|
88
|
+
force_initial_click: bool = False,
|
|
89
|
+
center_click_zoom_epsilon: float | None = 0.02,
|
|
90
|
+
always_refine: bool = False,
|
|
91
|
+
):
|
|
92
|
+
if max_levels < 1:
|
|
93
|
+
raise ValueError("max_levels must be at least 1")
|
|
94
|
+
if span_cells <= 1:
|
|
95
|
+
raise ValueError("span_cells must be greater than 1")
|
|
96
|
+
if center_click_zoom_epsilon is not None and not (
|
|
97
|
+
0 <= center_click_zoom_epsilon < 0.5
|
|
98
|
+
):
|
|
99
|
+
raise ValueError("center_click_zoom_epsilon must be in [0,0.5)")
|
|
100
|
+
self.provider = provider
|
|
101
|
+
self.max_levels = max_levels
|
|
102
|
+
self.span_cells = span_cells
|
|
103
|
+
self.force_initial_click = force_initial_click
|
|
104
|
+
self.center_click_zoom_epsilon = center_click_zoom_epsilon
|
|
105
|
+
self.always_refine = always_refine
|
|
106
|
+
self.last_trace: AdaptiveZoomTrace | None = None
|
|
107
|
+
self.last_zoom_images: list[tuple[Image.Image, Image.Image]] = []
|
|
108
|
+
self.last_metadata = None
|
|
109
|
+
|
|
110
|
+
def choose_action(
|
|
111
|
+
self,
|
|
112
|
+
*,
|
|
113
|
+
task: str,
|
|
114
|
+
clean: Image.Image,
|
|
115
|
+
gridded: Image.Image,
|
|
116
|
+
grid: GridSpec,
|
|
117
|
+
step: int,
|
|
118
|
+
) -> Action:
|
|
119
|
+
original = clean
|
|
120
|
+
full_box = (0, 0, clean.width, clean.height)
|
|
121
|
+
view_box = full_box
|
|
122
|
+
view_clean = clean
|
|
123
|
+
view_grid = gridded
|
|
124
|
+
levels = []
|
|
125
|
+
call_metadata = []
|
|
126
|
+
self.last_zoom_images = []
|
|
127
|
+
|
|
128
|
+
for level in range(1, self.max_levels + 1):
|
|
129
|
+
level_task = task
|
|
130
|
+
if level > 1:
|
|
131
|
+
level_task += (
|
|
132
|
+
f"\nAdaptive zoom level {level}: this is an enlarged crop from "
|
|
133
|
+
"the previous requested zoom. Re-evaluate whether the target can "
|
|
134
|
+
"now be clicked reliably."
|
|
135
|
+
)
|
|
136
|
+
chooser = (
|
|
137
|
+
self.provider.choose_action
|
|
138
|
+
if self.always_refine or level == self.max_levels
|
|
139
|
+
else self.provider.choose_action_or_zoom
|
|
140
|
+
)
|
|
141
|
+
decision = chooser(
|
|
142
|
+
task=level_task,
|
|
143
|
+
clean=view_clean,
|
|
144
|
+
gridded=view_grid,
|
|
145
|
+
grid=grid,
|
|
146
|
+
step=step,
|
|
147
|
+
force_click=self.force_initial_click or level > 1,
|
|
148
|
+
)
|
|
149
|
+
metadata = getattr(self.provider, "last_metadata", None)
|
|
150
|
+
call_metadata.append(metadata)
|
|
151
|
+
|
|
152
|
+
if (
|
|
153
|
+
not isinstance(decision, dict)
|
|
154
|
+
and decision.type == "click"
|
|
155
|
+
and level < self.max_levels
|
|
156
|
+
and not (metadata or {}).get("ocr_direct", False)
|
|
157
|
+
and (
|
|
158
|
+
self.always_refine
|
|
159
|
+
or (
|
|
160
|
+
self.center_click_zoom_epsilon is not None
|
|
161
|
+
and abs(decision.target.offset_x - 0.5)
|
|
162
|
+
<= self.center_click_zoom_epsilon
|
|
163
|
+
and abs(decision.target.offset_y - 0.5)
|
|
164
|
+
<= self.center_click_zoom_epsilon
|
|
165
|
+
)
|
|
166
|
+
)
|
|
167
|
+
):
|
|
168
|
+
decision = {
|
|
169
|
+
"type": "zoom",
|
|
170
|
+
"cell": decision.target.cell,
|
|
171
|
+
"reason": (
|
|
172
|
+
"fixed-refinement"
|
|
173
|
+
if self.always_refine
|
|
174
|
+
else "model-returned-suspicious-cell-center"
|
|
175
|
+
),
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
if isinstance(decision, dict):
|
|
179
|
+
display_crop = neighborhood_crop_box(
|
|
180
|
+
view_clean.size,
|
|
181
|
+
grid,
|
|
182
|
+
decision["cell"],
|
|
183
|
+
span_cells=self.span_cells,
|
|
184
|
+
)
|
|
185
|
+
next_view_box = _map_crop_to_original(
|
|
186
|
+
display_crop, view_box, view_clean.size
|
|
187
|
+
)
|
|
188
|
+
levels.append(
|
|
189
|
+
AdaptiveZoomLevel(
|
|
190
|
+
level,
|
|
191
|
+
view_box,
|
|
192
|
+
dict(decision),
|
|
193
|
+
next_view_box,
|
|
194
|
+
metadata,
|
|
195
|
+
)
|
|
196
|
+
)
|
|
197
|
+
view_box = next_view_box
|
|
198
|
+
view_clean = original.crop(view_box).resize(
|
|
199
|
+
original.size, Image.Resampling.LANCZOS
|
|
200
|
+
)
|
|
201
|
+
view_grid = render_numbered_grid(view_clean, grid)
|
|
202
|
+
self.last_zoom_images.append((view_clean, view_grid))
|
|
203
|
+
continue
|
|
204
|
+
|
|
205
|
+
dumped = decision.model_dump(mode="json")
|
|
206
|
+
if decision.type == "done":
|
|
207
|
+
levels.append(
|
|
208
|
+
AdaptiveZoomLevel(level, view_box, dumped, None, metadata)
|
|
209
|
+
)
|
|
210
|
+
final = decision
|
|
211
|
+
elif decision.type == "click":
|
|
212
|
+
local_mapper = GridMapper(
|
|
213
|
+
Region(width=view_clean.width, height=view_clean.height), grid
|
|
214
|
+
)
|
|
215
|
+
local_point = local_mapper.to_screen(decision.target)
|
|
216
|
+
full_x, full_y = _map_point_to_original(
|
|
217
|
+
local_point, view_box, view_clean.size
|
|
218
|
+
)
|
|
219
|
+
full_mapper = GridMapper(
|
|
220
|
+
Region(width=original.width, height=original.height), grid
|
|
221
|
+
)
|
|
222
|
+
final_point = full_mapper.from_screen(full_x, full_y)
|
|
223
|
+
final = parse_action(
|
|
224
|
+
{"type": "click", "target": final_point.model_dump(mode="json")}
|
|
225
|
+
)
|
|
226
|
+
levels.append(
|
|
227
|
+
AdaptiveZoomLevel(level, view_box, dumped, None, metadata)
|
|
228
|
+
)
|
|
229
|
+
else:
|
|
230
|
+
raise RuntimeError("adaptive zoom currently supports click or done")
|
|
231
|
+
|
|
232
|
+
self.last_trace = AdaptiveZoomTrace(
|
|
233
|
+
tuple(levels), final.model_dump(mode="json")
|
|
234
|
+
)
|
|
235
|
+
self.last_metadata = {
|
|
236
|
+
"adaptive_zoom_levels": level,
|
|
237
|
+
"calls": call_metadata,
|
|
238
|
+
"total_elapsed_seconds": round(
|
|
239
|
+
sum((item or {}).get("elapsed_seconds", 0) for item in call_metadata),
|
|
240
|
+
3,
|
|
241
|
+
),
|
|
242
|
+
}
|
|
243
|
+
return final
|
|
244
|
+
|
|
245
|
+
raise AssertionError("adaptive zoom loop ended without a decision")
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
import Foundation
|
|
2
|
+
import ImageIO
|
|
3
|
+
import Vision
|
|
4
|
+
|
|
5
|
+
guard CommandLine.arguments.count >= 3 else {
|
|
6
|
+
fputs("usage: apple_vision_ocr.swift IMAGE fast|accurate [WORDS] [LANGUAGES]\n", stderr)
|
|
7
|
+
exit(2)
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
let path = CommandLine.arguments[1]
|
|
11
|
+
let mode = CommandLine.arguments[2]
|
|
12
|
+
let customWords = CommandLine.arguments.count >= 4
|
|
13
|
+
? CommandLine.arguments[3].split(separator: ",").map(String.init)
|
|
14
|
+
: []
|
|
15
|
+
let languages = CommandLine.arguments.count >= 5
|
|
16
|
+
? CommandLine.arguments[4].split(separator: ",").map(String.init)
|
|
17
|
+
: ["zh-Hans", "en-US"]
|
|
18
|
+
|
|
19
|
+
guard mode == "fast" || mode == "accurate" else {
|
|
20
|
+
fputs("mode must be fast or accurate\n", stderr)
|
|
21
|
+
exit(2)
|
|
22
|
+
}
|
|
23
|
+
guard let source = CGImageSourceCreateWithURL(URL(fileURLWithPath: path) as CFURL, nil),
|
|
24
|
+
let image = CGImageSourceCreateImageAtIndex(source, 0, nil) else {
|
|
25
|
+
fputs("cannot load image\n", stderr)
|
|
26
|
+
exit(2)
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
let request = VNRecognizeTextRequest()
|
|
30
|
+
request.recognitionLevel = mode == "accurate" ? .accurate : .fast
|
|
31
|
+
request.recognitionLanguages = languages
|
|
32
|
+
request.usesLanguageCorrection = false
|
|
33
|
+
request.minimumTextHeight = 0.005
|
|
34
|
+
request.customWords = customWords
|
|
35
|
+
|
|
36
|
+
let started = CFAbsoluteTimeGetCurrent()
|
|
37
|
+
do {
|
|
38
|
+
try VNImageRequestHandler(cgImage: image, options: [:]).perform([request])
|
|
39
|
+
} catch {
|
|
40
|
+
fputs("Vision request failed: \(error)\n", stderr)
|
|
41
|
+
exit(1)
|
|
42
|
+
}
|
|
43
|
+
let elapsedMs = (CFAbsoluteTimeGetCurrent() - started) * 1000
|
|
44
|
+
|
|
45
|
+
let width = CGFloat(image.width)
|
|
46
|
+
let height = CGFloat(image.height)
|
|
47
|
+
let rows: [[String: Any]] = (request.results ?? []).compactMap { observation in
|
|
48
|
+
guard let candidate = observation.topCandidates(1).first else { return nil }
|
|
49
|
+
let box = observation.boundingBox
|
|
50
|
+
return [
|
|
51
|
+
"text": candidate.string,
|
|
52
|
+
"confidence": Double(candidate.confidence),
|
|
53
|
+
"bbox": [
|
|
54
|
+
Double(box.minX * width),
|
|
55
|
+
Double((1 - box.maxY) * height),
|
|
56
|
+
Double(box.maxX * width),
|
|
57
|
+
Double((1 - box.minY) * height),
|
|
58
|
+
],
|
|
59
|
+
]
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
let output: [String: Any] = [
|
|
63
|
+
"mode": mode,
|
|
64
|
+
"elapsed_ms": elapsedMs,
|
|
65
|
+
"image_size": [image.width, image.height],
|
|
66
|
+
"results": rows,
|
|
67
|
+
]
|
|
68
|
+
let data = try JSONSerialization.data(
|
|
69
|
+
withJSONObject: output,
|
|
70
|
+
options: [.prettyPrinted, .sortedKeys]
|
|
71
|
+
)
|
|
72
|
+
FileHandle.standardOutput.write(data)
|
|
73
|
+
FileHandle.standardOutput.write(Data("\n".utf8))
|
vco/browser.py
ADDED
|
@@ -0,0 +1,341 @@
|
|
|
1
|
+
"""Headless page rendering and DOM interaction via Playwright."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class _PageLog:
|
|
9
|
+
"""Collect console errors, uncaught exceptions, and failed requests."""
|
|
10
|
+
|
|
11
|
+
def __init__(self, limit: int = 20):
|
|
12
|
+
self.console_errors: list[str] = []
|
|
13
|
+
self.page_errors: list[str] = []
|
|
14
|
+
self.failed_requests: list[str] = []
|
|
15
|
+
self._limit = limit
|
|
16
|
+
|
|
17
|
+
def attach(self, page) -> None:
|
|
18
|
+
def on_console(message):
|
|
19
|
+
if message.type == "error" and len(self.console_errors) < self._limit:
|
|
20
|
+
self.console_errors.append(message.text)
|
|
21
|
+
|
|
22
|
+
def on_page_error(error):
|
|
23
|
+
if len(self.page_errors) < self._limit:
|
|
24
|
+
self.page_errors.append(str(error))
|
|
25
|
+
|
|
26
|
+
def on_request_failed(request):
|
|
27
|
+
if len(self.failed_requests) < self._limit:
|
|
28
|
+
failure = request.failure or ""
|
|
29
|
+
self.failed_requests.append(f"{request.method} {request.url} {failure}")
|
|
30
|
+
|
|
31
|
+
page.on("console", on_console)
|
|
32
|
+
page.on("pageerror", on_page_error)
|
|
33
|
+
page.on("requestfailed", on_request_failed)
|
|
34
|
+
|
|
35
|
+
def as_dict(self) -> dict:
|
|
36
|
+
return {
|
|
37
|
+
"console_errors": self.console_errors,
|
|
38
|
+
"page_errors": self.page_errors,
|
|
39
|
+
"failed_requests": self.failed_requests,
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _open(pw, url, *, width, height, timeout, profile=None, log=None, headless=True,
|
|
44
|
+
record_dir=None):
|
|
45
|
+
"""Launch (persistent when ``profile`` is set) and open ``url``."""
|
|
46
|
+
video = {}
|
|
47
|
+
if record_dir:
|
|
48
|
+
video["record_video_dir"] = str(record_dir)
|
|
49
|
+
video["record_video_size"] = {"width": width, "height": height}
|
|
50
|
+
if profile:
|
|
51
|
+
context = pw.chromium.launch_persistent_context(
|
|
52
|
+
profile,
|
|
53
|
+
headless=headless,
|
|
54
|
+
viewport={"width": width, "height": height},
|
|
55
|
+
**video,
|
|
56
|
+
)
|
|
57
|
+
browser = None
|
|
58
|
+
else:
|
|
59
|
+
browser = pw.chromium.launch(headless=headless)
|
|
60
|
+
context = browser.new_context(
|
|
61
|
+
viewport={"width": width, "height": height}, **video
|
|
62
|
+
)
|
|
63
|
+
page = context.pages[0] if context.pages else context.new_page()
|
|
64
|
+
if log is not None:
|
|
65
|
+
log.attach(page)
|
|
66
|
+
page.goto(url, timeout=timeout * 1000)
|
|
67
|
+
page.wait_for_load_state("load", timeout=timeout * 1000)
|
|
68
|
+
return browser, context, page
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _close(browser, context, page=None, video_path: str | None = None) -> str | None:
|
|
72
|
+
"""Close and, when recording, move the video to ``video_path``."""
|
|
73
|
+
context.close()
|
|
74
|
+
saved = None
|
|
75
|
+
if page is not None and video_path is not None and page.video is not None:
|
|
76
|
+
original = page.video.path()
|
|
77
|
+
page.video.save_as(video_path)
|
|
78
|
+
Path(original).unlink(missing_ok=True)
|
|
79
|
+
saved = video_path
|
|
80
|
+
if browser is not None:
|
|
81
|
+
browser.close()
|
|
82
|
+
return saved
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _load_pw():
|
|
86
|
+
try:
|
|
87
|
+
from playwright.sync_api import sync_playwright
|
|
88
|
+
except ImportError as exc:
|
|
89
|
+
raise RuntimeError("requires: pip install playwright") from exc
|
|
90
|
+
return sync_playwright
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def screenshot(
|
|
94
|
+
url: str,
|
|
95
|
+
output: str,
|
|
96
|
+
*,
|
|
97
|
+
full_page: bool = False,
|
|
98
|
+
width: int = 1280,
|
|
99
|
+
height: int = 800,
|
|
100
|
+
timeout: float = 15.0,
|
|
101
|
+
settle: float = 0.0,
|
|
102
|
+
profile: str | None = None,
|
|
103
|
+
headless: bool = True,
|
|
104
|
+
hold: float = 0.0,
|
|
105
|
+
record: str | None = None,
|
|
106
|
+
) -> dict:
|
|
107
|
+
"""Render ``url`` in Chromium and save a screenshot."""
|
|
108
|
+
sync_playwright = _load_pw()
|
|
109
|
+
log = _PageLog()
|
|
110
|
+
with sync_playwright() as pw:
|
|
111
|
+
record_dir = None if record is None else str(Path(record).parent)
|
|
112
|
+
browser, context, page = _open(
|
|
113
|
+
pw, url, width=width, height=height, timeout=timeout, profile=profile,
|
|
114
|
+
log=log, headless=headless, record_dir=record_dir,
|
|
115
|
+
)
|
|
116
|
+
try:
|
|
117
|
+
if settle > 0:
|
|
118
|
+
page.wait_for_timeout(int(settle * 1000))
|
|
119
|
+
page.screenshot(path=output, full_page=full_page)
|
|
120
|
+
result = {
|
|
121
|
+
"path": output,
|
|
122
|
+
"url": page.url,
|
|
123
|
+
"title": page.title(),
|
|
124
|
+
"viewport": {"width": width, "height": height},
|
|
125
|
+
"full_page": full_page,
|
|
126
|
+
"headless": headless,
|
|
127
|
+
}
|
|
128
|
+
result.update(log.as_dict())
|
|
129
|
+
if hold > 0:
|
|
130
|
+
page.wait_for_timeout(int(hold * 1000))
|
|
131
|
+
return result
|
|
132
|
+
finally:
|
|
133
|
+
video = _close(browser, context, page=page, video_path=record)
|
|
134
|
+
if video:
|
|
135
|
+
result["video"] = video
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _apply_fills(page, fills: list[str], visible: bool = False) -> list[str]:
|
|
139
|
+
"""Fill inputs as ``placeholder=text`` pairs; returns error strings.
|
|
140
|
+
|
|
141
|
+
With ``visible=True`` types character by character so a human watching a
|
|
142
|
+
headed browser can see the text appear.
|
|
143
|
+
"""
|
|
144
|
+
errors = []
|
|
145
|
+
for item in fills:
|
|
146
|
+
if "=" not in item:
|
|
147
|
+
errors.append(f"invalid --fill {item!r}; expected placeholder=text")
|
|
148
|
+
continue
|
|
149
|
+
placeholder, text = item.split("=", 1)
|
|
150
|
+
locator = page.get_by_placeholder(placeholder, exact=True)
|
|
151
|
+
if locator.count() == 0:
|
|
152
|
+
locator = page.get_by_placeholder(placeholder)
|
|
153
|
+
if locator.count() != 1:
|
|
154
|
+
errors.append(
|
|
155
|
+
f"--fill {placeholder!r}: expected 1 input, got {locator.count()}"
|
|
156
|
+
)
|
|
157
|
+
continue
|
|
158
|
+
if visible:
|
|
159
|
+
locator.first.click()
|
|
160
|
+
locator.first.press_sequentially(text, delay=90)
|
|
161
|
+
else:
|
|
162
|
+
locator.first.fill(text)
|
|
163
|
+
return errors
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _flash_ring(page, cx: float, cy: float, radius: float) -> None:
|
|
167
|
+
"""Draw a temporary orange halo at (cx, cy): transparent core, dense rim."""
|
|
168
|
+
page.evaluate(
|
|
169
|
+
"""([x, y, r]) => {
|
|
170
|
+
const el = document.createElement('div');
|
|
171
|
+
el.style.cssText = 'position:fixed;pointer-events:none;z-index:2147483647;'
|
|
172
|
+
+ 'left:' + (x - r) + 'px;top:' + (y - r) + 'px;'
|
|
173
|
+
+ 'width:' + (2 * r) + 'px;height:' + (2 * r) + 'px;'
|
|
174
|
+
+ 'border-radius:50%;'
|
|
175
|
+
+ 'background:radial-gradient(circle,'
|
|
176
|
+
+ ' rgba(255,140,0,0) 30%, rgba(255,140,0,0.85) 55%,'
|
|
177
|
+
+ ' rgba(255,140,0,0.30) 75%, rgba(255,140,0,0) 100%);';
|
|
178
|
+
document.body.appendChild(el);
|
|
179
|
+
setTimeout(() => el.remove(), 1400);
|
|
180
|
+
}""",
|
|
181
|
+
[cx, cy, radius],
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def click(
|
|
186
|
+
url: str,
|
|
187
|
+
target: str | None,
|
|
188
|
+
*,
|
|
189
|
+
selector: str | None = None,
|
|
190
|
+
contains: bool = False,
|
|
191
|
+
fills: list[str] | None = None,
|
|
192
|
+
width: int = 1280,
|
|
193
|
+
height: int = 800,
|
|
194
|
+
timeout: float = 15.0,
|
|
195
|
+
settle: float = 0.5,
|
|
196
|
+
profile: str | None = None,
|
|
197
|
+
headless: bool = True,
|
|
198
|
+
hold: float = 0.0,
|
|
199
|
+
expect: str | None = None,
|
|
200
|
+
expect_timeout: float = 10.0,
|
|
201
|
+
record: str | None = None,
|
|
202
|
+
before_path: str | None = None,
|
|
203
|
+
after_path: str | None = None,
|
|
204
|
+
) -> dict:
|
|
205
|
+
"""Open ``url``, optionally fill inputs, DOM-click a unique ``target``."""
|
|
206
|
+
sync_playwright = _load_pw()
|
|
207
|
+
log = _PageLog()
|
|
208
|
+
with sync_playwright() as pw:
|
|
209
|
+
record_dir = None if record is None else str(Path(record).parent)
|
|
210
|
+
browser, context, page = _open(
|
|
211
|
+
pw, url, width=width, height=height, timeout=timeout, profile=profile,
|
|
212
|
+
log=log, headless=headless, record_dir=record_dir,
|
|
213
|
+
)
|
|
214
|
+
try:
|
|
215
|
+
if before_path:
|
|
216
|
+
page.screenshot(path=before_path, full_page=False)
|
|
217
|
+
|
|
218
|
+
demo = not headless or record is not None
|
|
219
|
+
fill_errors = _apply_fills(page, fills or [], visible=demo)
|
|
220
|
+
|
|
221
|
+
if selector:
|
|
222
|
+
locator = page.locator(selector)
|
|
223
|
+
else:
|
|
224
|
+
locator = page.get_by_text(target, exact=True)
|
|
225
|
+
count = locator.count()
|
|
226
|
+
if count == 0 and not selector and contains:
|
|
227
|
+
locator = page.get_by_text(target)
|
|
228
|
+
count = locator.count()
|
|
229
|
+
candidates = []
|
|
230
|
+
for i in range(min(count, 8)):
|
|
231
|
+
item = locator.nth(i)
|
|
232
|
+
box = item.bounding_box()
|
|
233
|
+
candidates.append(
|
|
234
|
+
{
|
|
235
|
+
"text": (item.text_content() or "").strip(),
|
|
236
|
+
"bbox": (
|
|
237
|
+
None
|
|
238
|
+
if box is None
|
|
239
|
+
else [
|
|
240
|
+
box["x"],
|
|
241
|
+
box["y"],
|
|
242
|
+
box["x"] + box["width"],
|
|
243
|
+
box["y"] + box["height"],
|
|
244
|
+
]
|
|
245
|
+
),
|
|
246
|
+
}
|
|
247
|
+
)
|
|
248
|
+
result = {
|
|
249
|
+
"clicked": False,
|
|
250
|
+
"url": page.url,
|
|
251
|
+
"target": selector or target,
|
|
252
|
+
"candidate_count": count,
|
|
253
|
+
"candidates": candidates,
|
|
254
|
+
"fill_errors": fill_errors,
|
|
255
|
+
"before": before_path,
|
|
256
|
+
"after": None,
|
|
257
|
+
"x": None,
|
|
258
|
+
"y": None,
|
|
259
|
+
}
|
|
260
|
+
if count != 1:
|
|
261
|
+
result["error"] = "no match" if count == 0 else f"ambiguous: {count} matches"
|
|
262
|
+
result.update(log.as_dict())
|
|
263
|
+
return result
|
|
264
|
+
box = locator.first.bounding_box()
|
|
265
|
+
if demo and box is not None:
|
|
266
|
+
page.wait_for_timeout(400)
|
|
267
|
+
_flash_ring(
|
|
268
|
+
page,
|
|
269
|
+
box["x"] + box["width"] / 2,
|
|
270
|
+
box["y"] + box["height"] / 2,
|
|
271
|
+
max(14.0, min(32.0, float(min(box["width"], box["height"])))),
|
|
272
|
+
)
|
|
273
|
+
page.wait_for_timeout(900)
|
|
274
|
+
locator.first.click()
|
|
275
|
+
page.wait_for_timeout(int(settle * 1000))
|
|
276
|
+
verified = None
|
|
277
|
+
if expect:
|
|
278
|
+
try:
|
|
279
|
+
page.get_by_text(expect).first.wait_for(
|
|
280
|
+
state="visible", timeout=int(expect_timeout * 1000)
|
|
281
|
+
)
|
|
282
|
+
verified = True
|
|
283
|
+
except Exception:
|
|
284
|
+
verified = False
|
|
285
|
+
if after_path:
|
|
286
|
+
page.screenshot(path=after_path, full_page=False)
|
|
287
|
+
result.update(
|
|
288
|
+
{
|
|
289
|
+
"clicked": True,
|
|
290
|
+
"x": None if box is None else round(box["x"] + box["width"] / 2),
|
|
291
|
+
"y": None if box is None else round(box["y"] + box["height"] / 2),
|
|
292
|
+
"after": after_path,
|
|
293
|
+
"expect": expect,
|
|
294
|
+
"verified": verified,
|
|
295
|
+
}
|
|
296
|
+
)
|
|
297
|
+
result.update(log.as_dict())
|
|
298
|
+
if hold > 0:
|
|
299
|
+
page.wait_for_timeout(int(hold * 1000))
|
|
300
|
+
return result
|
|
301
|
+
finally:
|
|
302
|
+
video = _close(browser, context, page=page, video_path=record)
|
|
303
|
+
if video:
|
|
304
|
+
result["video"] = video
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def snapshot(
|
|
308
|
+
url: str,
|
|
309
|
+
*,
|
|
310
|
+
width: int = 1280,
|
|
311
|
+
height: int = 800,
|
|
312
|
+
timeout: float = 15.0,
|
|
313
|
+
settle: float = 0.0,
|
|
314
|
+
profile: str | None = None,
|
|
315
|
+
headless: bool = True,
|
|
316
|
+
hold: float = 0.0,
|
|
317
|
+
) -> dict:
|
|
318
|
+
"""Return the page's accessibility tree as text (no vision model needed)."""
|
|
319
|
+
sync_playwright = _load_pw()
|
|
320
|
+
log = _PageLog()
|
|
321
|
+
with sync_playwright() as pw:
|
|
322
|
+
browser, context, page = _open(
|
|
323
|
+
pw, url, width=width, height=height, timeout=timeout, profile=profile,
|
|
324
|
+
log=log, headless=headless,
|
|
325
|
+
)
|
|
326
|
+
try:
|
|
327
|
+
if settle > 0:
|
|
328
|
+
page.wait_for_timeout(int(settle * 1000))
|
|
329
|
+
tree = page.locator("body").aria_snapshot()
|
|
330
|
+
result = {
|
|
331
|
+
"url": page.url,
|
|
332
|
+
"title": page.title(),
|
|
333
|
+
"snapshot": tree,
|
|
334
|
+
"headless": headless,
|
|
335
|
+
}
|
|
336
|
+
result.update(log.as_dict())
|
|
337
|
+
if hold > 0:
|
|
338
|
+
page.wait_for_timeout(int(hold * 1000))
|
|
339
|
+
return result
|
|
340
|
+
finally:
|
|
341
|
+
_close(browser, context)
|