vco 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
vco/__init__.py ADDED
@@ -0,0 +1,35 @@
1
+ """vco — a clicker for LLMs: operate web pages and desktop screens."""
2
+
3
+ from .geometry import GridMapper
4
+ from .models import Action, GridPoint, GridSpec, Region, parse_action
5
+ from .ocr import (
6
+ HTTPOCRBackend,
7
+ OCRBackend,
8
+ OCRBox,
9
+ OCRRequest,
10
+ OCRResult,
11
+ RapidOCRBackend,
12
+ create_ocr_backend,
13
+ register_ocr_backend,
14
+ )
15
+ from .ocr_assist import OCRAssistProvider
16
+
17
+ __all__ = [
18
+ "Action",
19
+ "GridMapper",
20
+ "GridPoint",
21
+ "GridSpec",
22
+ "OCRBackend",
23
+ "OCRAssistProvider",
24
+ "OCRBox",
25
+ "OCRRequest",
26
+ "OCRResult",
27
+ "HTTPOCRBackend",
28
+ "RapidOCRBackend",
29
+ "Region",
30
+ "create_ocr_backend",
31
+ "parse_action",
32
+ "register_ocr_backend",
33
+ ]
34
+
35
+ __version__ = "0.1.0"
vco/adaptive_zoom.py ADDED
@@ -0,0 +1,245 @@
1
+ """Model-directed recursive zoom for dense-grid visual localization."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import math
6
+ from dataclasses import dataclass
7
+ from typing import Tuple
8
+
9
+ from PIL import Image
10
+
11
+ from .geometry import GridMapper
12
+ from .grid import render_numbered_grid
13
+ from .models import Action, GridSpec, Region, parse_action
14
+ from .zoom import neighborhood_crop_box
15
+
16
+
17
+ Box = Tuple[int, int, int, int]
18
+
19
+
20
+ @dataclass(frozen=True)
21
+ class AdaptiveZoomLevel:
22
+ level: int
23
+ view_box: Box
24
+ decision: dict
25
+ next_view_box: Box | None
26
+ provider_metadata: dict | None
27
+
28
+ def as_dict(self) -> dict:
29
+ return {
30
+ "level": self.level,
31
+ "view_box_image_pixels": self.view_box,
32
+ "decision": self.decision,
33
+ "next_view_box_image_pixels": self.next_view_box,
34
+ "provider_metadata": self.provider_metadata,
35
+ }
36
+
37
+
38
+ @dataclass(frozen=True)
39
+ class AdaptiveZoomTrace:
40
+ levels: tuple[AdaptiveZoomLevel, ...]
41
+ final_action: dict
42
+
43
+ def as_dict(self) -> dict:
44
+ return {
45
+ "mode": "model-directed-adaptive-zoom",
46
+ "levels": [level.as_dict() for level in self.levels],
47
+ "final_action": self.final_action,
48
+ }
49
+
50
+
51
+ def _map_crop_to_original(display_crop: Box, view_box: Box, display_size) -> Box:
52
+ display_width, display_height = display_size
53
+ view_left, view_top, view_right, view_bottom = view_box
54
+ view_width = view_right - view_left
55
+ view_height = view_bottom - view_top
56
+ left, top, right, bottom = display_crop
57
+ mapped = (
58
+ view_left + math.floor(left * view_width / display_width),
59
+ view_top + math.floor(top * view_height / display_height),
60
+ view_left + math.ceil(right * view_width / display_width),
61
+ view_top + math.ceil(bottom * view_height / display_height),
62
+ )
63
+ if mapped[2] <= mapped[0] or mapped[3] <= mapped[1]:
64
+ raise ValueError("adaptive zoom produced an empty crop")
65
+ return mapped
66
+
67
+
68
+ def _map_point_to_original(point, view_box: Box, display_size):
69
+ x, y = point
70
+ display_width, display_height = display_size
71
+ left, top, right, bottom = view_box
72
+ view_width = right - left
73
+ view_height = bottom - top
74
+ original_x = left + round(x * max(0, view_width - 1) / max(1, display_width - 1))
75
+ original_y = top + round(y * max(0, view_height - 1) / max(1, display_height - 1))
76
+ return original_x, original_y
77
+
78
+
79
+ class AdaptiveZoomProvider:
80
+ """Ask the model to click now or request another zoom level."""
81
+
82
+ def __init__(
83
+ self,
84
+ provider,
85
+ *,
86
+ max_levels: int = 3,
87
+ span_cells: float = 4.0,
88
+ force_initial_click: bool = False,
89
+ center_click_zoom_epsilon: float | None = 0.02,
90
+ always_refine: bool = False,
91
+ ):
92
+ if max_levels < 1:
93
+ raise ValueError("max_levels must be at least 1")
94
+ if span_cells <= 1:
95
+ raise ValueError("span_cells must be greater than 1")
96
+ if center_click_zoom_epsilon is not None and not (
97
+ 0 <= center_click_zoom_epsilon < 0.5
98
+ ):
99
+ raise ValueError("center_click_zoom_epsilon must be in [0,0.5)")
100
+ self.provider = provider
101
+ self.max_levels = max_levels
102
+ self.span_cells = span_cells
103
+ self.force_initial_click = force_initial_click
104
+ self.center_click_zoom_epsilon = center_click_zoom_epsilon
105
+ self.always_refine = always_refine
106
+ self.last_trace: AdaptiveZoomTrace | None = None
107
+ self.last_zoom_images: list[tuple[Image.Image, Image.Image]] = []
108
+ self.last_metadata = None
109
+
110
+ def choose_action(
111
+ self,
112
+ *,
113
+ task: str,
114
+ clean: Image.Image,
115
+ gridded: Image.Image,
116
+ grid: GridSpec,
117
+ step: int,
118
+ ) -> Action:
119
+ original = clean
120
+ full_box = (0, 0, clean.width, clean.height)
121
+ view_box = full_box
122
+ view_clean = clean
123
+ view_grid = gridded
124
+ levels = []
125
+ call_metadata = []
126
+ self.last_zoom_images = []
127
+
128
+ for level in range(1, self.max_levels + 1):
129
+ level_task = task
130
+ if level > 1:
131
+ level_task += (
132
+ f"\nAdaptive zoom level {level}: this is an enlarged crop from "
133
+ "the previous requested zoom. Re-evaluate whether the target can "
134
+ "now be clicked reliably."
135
+ )
136
+ chooser = (
137
+ self.provider.choose_action
138
+ if self.always_refine or level == self.max_levels
139
+ else self.provider.choose_action_or_zoom
140
+ )
141
+ decision = chooser(
142
+ task=level_task,
143
+ clean=view_clean,
144
+ gridded=view_grid,
145
+ grid=grid,
146
+ step=step,
147
+ force_click=self.force_initial_click or level > 1,
148
+ )
149
+ metadata = getattr(self.provider, "last_metadata", None)
150
+ call_metadata.append(metadata)
151
+
152
+ if (
153
+ not isinstance(decision, dict)
154
+ and decision.type == "click"
155
+ and level < self.max_levels
156
+ and not (metadata or {}).get("ocr_direct", False)
157
+ and (
158
+ self.always_refine
159
+ or (
160
+ self.center_click_zoom_epsilon is not None
161
+ and abs(decision.target.offset_x - 0.5)
162
+ <= self.center_click_zoom_epsilon
163
+ and abs(decision.target.offset_y - 0.5)
164
+ <= self.center_click_zoom_epsilon
165
+ )
166
+ )
167
+ ):
168
+ decision = {
169
+ "type": "zoom",
170
+ "cell": decision.target.cell,
171
+ "reason": (
172
+ "fixed-refinement"
173
+ if self.always_refine
174
+ else "model-returned-suspicious-cell-center"
175
+ ),
176
+ }
177
+
178
+ if isinstance(decision, dict):
179
+ display_crop = neighborhood_crop_box(
180
+ view_clean.size,
181
+ grid,
182
+ decision["cell"],
183
+ span_cells=self.span_cells,
184
+ )
185
+ next_view_box = _map_crop_to_original(
186
+ display_crop, view_box, view_clean.size
187
+ )
188
+ levels.append(
189
+ AdaptiveZoomLevel(
190
+ level,
191
+ view_box,
192
+ dict(decision),
193
+ next_view_box,
194
+ metadata,
195
+ )
196
+ )
197
+ view_box = next_view_box
198
+ view_clean = original.crop(view_box).resize(
199
+ original.size, Image.Resampling.LANCZOS
200
+ )
201
+ view_grid = render_numbered_grid(view_clean, grid)
202
+ self.last_zoom_images.append((view_clean, view_grid))
203
+ continue
204
+
205
+ dumped = decision.model_dump(mode="json")
206
+ if decision.type == "done":
207
+ levels.append(
208
+ AdaptiveZoomLevel(level, view_box, dumped, None, metadata)
209
+ )
210
+ final = decision
211
+ elif decision.type == "click":
212
+ local_mapper = GridMapper(
213
+ Region(width=view_clean.width, height=view_clean.height), grid
214
+ )
215
+ local_point = local_mapper.to_screen(decision.target)
216
+ full_x, full_y = _map_point_to_original(
217
+ local_point, view_box, view_clean.size
218
+ )
219
+ full_mapper = GridMapper(
220
+ Region(width=original.width, height=original.height), grid
221
+ )
222
+ final_point = full_mapper.from_screen(full_x, full_y)
223
+ final = parse_action(
224
+ {"type": "click", "target": final_point.model_dump(mode="json")}
225
+ )
226
+ levels.append(
227
+ AdaptiveZoomLevel(level, view_box, dumped, None, metadata)
228
+ )
229
+ else:
230
+ raise RuntimeError("adaptive zoom currently supports click or done")
231
+
232
+ self.last_trace = AdaptiveZoomTrace(
233
+ tuple(levels), final.model_dump(mode="json")
234
+ )
235
+ self.last_metadata = {
236
+ "adaptive_zoom_levels": level,
237
+ "calls": call_metadata,
238
+ "total_elapsed_seconds": round(
239
+ sum((item or {}).get("elapsed_seconds", 0) for item in call_metadata),
240
+ 3,
241
+ ),
242
+ }
243
+ return final
244
+
245
+ raise AssertionError("adaptive zoom loop ended without a decision")
@@ -0,0 +1,73 @@
1
+ import Foundation
2
+ import ImageIO
3
+ import Vision
4
+
5
+ guard CommandLine.arguments.count >= 3 else {
6
+ fputs("usage: apple_vision_ocr.swift IMAGE fast|accurate [WORDS] [LANGUAGES]\n", stderr)
7
+ exit(2)
8
+ }
9
+
10
+ let path = CommandLine.arguments[1]
11
+ let mode = CommandLine.arguments[2]
12
+ let customWords = CommandLine.arguments.count >= 4
13
+ ? CommandLine.arguments[3].split(separator: ",").map(String.init)
14
+ : []
15
+ let languages = CommandLine.arguments.count >= 5
16
+ ? CommandLine.arguments[4].split(separator: ",").map(String.init)
17
+ : ["zh-Hans", "en-US"]
18
+
19
+ guard mode == "fast" || mode == "accurate" else {
20
+ fputs("mode must be fast or accurate\n", stderr)
21
+ exit(2)
22
+ }
23
+ guard let source = CGImageSourceCreateWithURL(URL(fileURLWithPath: path) as CFURL, nil),
24
+ let image = CGImageSourceCreateImageAtIndex(source, 0, nil) else {
25
+ fputs("cannot load image\n", stderr)
26
+ exit(2)
27
+ }
28
+
29
+ let request = VNRecognizeTextRequest()
30
+ request.recognitionLevel = mode == "accurate" ? .accurate : .fast
31
+ request.recognitionLanguages = languages
32
+ request.usesLanguageCorrection = false
33
+ request.minimumTextHeight = 0.005
34
+ request.customWords = customWords
35
+
36
+ let started = CFAbsoluteTimeGetCurrent()
37
+ do {
38
+ try VNImageRequestHandler(cgImage: image, options: [:]).perform([request])
39
+ } catch {
40
+ fputs("Vision request failed: \(error)\n", stderr)
41
+ exit(1)
42
+ }
43
+ let elapsedMs = (CFAbsoluteTimeGetCurrent() - started) * 1000
44
+
45
+ let width = CGFloat(image.width)
46
+ let height = CGFloat(image.height)
47
+ let rows: [[String: Any]] = (request.results ?? []).compactMap { observation in
48
+ guard let candidate = observation.topCandidates(1).first else { return nil }
49
+ let box = observation.boundingBox
50
+ return [
51
+ "text": candidate.string,
52
+ "confidence": Double(candidate.confidence),
53
+ "bbox": [
54
+ Double(box.minX * width),
55
+ Double((1 - box.maxY) * height),
56
+ Double(box.maxX * width),
57
+ Double((1 - box.minY) * height),
58
+ ],
59
+ ]
60
+ }
61
+
62
+ let output: [String: Any] = [
63
+ "mode": mode,
64
+ "elapsed_ms": elapsedMs,
65
+ "image_size": [image.width, image.height],
66
+ "results": rows,
67
+ ]
68
+ let data = try JSONSerialization.data(
69
+ withJSONObject: output,
70
+ options: [.prettyPrinted, .sortedKeys]
71
+ )
72
+ FileHandle.standardOutput.write(data)
73
+ FileHandle.standardOutput.write(Data("\n".utf8))
vco/browser.py ADDED
@@ -0,0 +1,341 @@
1
+ """Headless page rendering and DOM interaction via Playwright."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+
7
+
8
+ class _PageLog:
9
+ """Collect console errors, uncaught exceptions, and failed requests."""
10
+
11
+ def __init__(self, limit: int = 20):
12
+ self.console_errors: list[str] = []
13
+ self.page_errors: list[str] = []
14
+ self.failed_requests: list[str] = []
15
+ self._limit = limit
16
+
17
+ def attach(self, page) -> None:
18
+ def on_console(message):
19
+ if message.type == "error" and len(self.console_errors) < self._limit:
20
+ self.console_errors.append(message.text)
21
+
22
+ def on_page_error(error):
23
+ if len(self.page_errors) < self._limit:
24
+ self.page_errors.append(str(error))
25
+
26
+ def on_request_failed(request):
27
+ if len(self.failed_requests) < self._limit:
28
+ failure = request.failure or ""
29
+ self.failed_requests.append(f"{request.method} {request.url} {failure}")
30
+
31
+ page.on("console", on_console)
32
+ page.on("pageerror", on_page_error)
33
+ page.on("requestfailed", on_request_failed)
34
+
35
+ def as_dict(self) -> dict:
36
+ return {
37
+ "console_errors": self.console_errors,
38
+ "page_errors": self.page_errors,
39
+ "failed_requests": self.failed_requests,
40
+ }
41
+
42
+
43
+ def _open(pw, url, *, width, height, timeout, profile=None, log=None, headless=True,
44
+ record_dir=None):
45
+ """Launch (persistent when ``profile`` is set) and open ``url``."""
46
+ video = {}
47
+ if record_dir:
48
+ video["record_video_dir"] = str(record_dir)
49
+ video["record_video_size"] = {"width": width, "height": height}
50
+ if profile:
51
+ context = pw.chromium.launch_persistent_context(
52
+ profile,
53
+ headless=headless,
54
+ viewport={"width": width, "height": height},
55
+ **video,
56
+ )
57
+ browser = None
58
+ else:
59
+ browser = pw.chromium.launch(headless=headless)
60
+ context = browser.new_context(
61
+ viewport={"width": width, "height": height}, **video
62
+ )
63
+ page = context.pages[0] if context.pages else context.new_page()
64
+ if log is not None:
65
+ log.attach(page)
66
+ page.goto(url, timeout=timeout * 1000)
67
+ page.wait_for_load_state("load", timeout=timeout * 1000)
68
+ return browser, context, page
69
+
70
+
71
+ def _close(browser, context, page=None, video_path: str | None = None) -> str | None:
72
+ """Close and, when recording, move the video to ``video_path``."""
73
+ context.close()
74
+ saved = None
75
+ if page is not None and video_path is not None and page.video is not None:
76
+ original = page.video.path()
77
+ page.video.save_as(video_path)
78
+ Path(original).unlink(missing_ok=True)
79
+ saved = video_path
80
+ if browser is not None:
81
+ browser.close()
82
+ return saved
83
+
84
+
85
+ def _load_pw():
86
+ try:
87
+ from playwright.sync_api import sync_playwright
88
+ except ImportError as exc:
89
+ raise RuntimeError("requires: pip install playwright") from exc
90
+ return sync_playwright
91
+
92
+
93
+ def screenshot(
94
+ url: str,
95
+ output: str,
96
+ *,
97
+ full_page: bool = False,
98
+ width: int = 1280,
99
+ height: int = 800,
100
+ timeout: float = 15.0,
101
+ settle: float = 0.0,
102
+ profile: str | None = None,
103
+ headless: bool = True,
104
+ hold: float = 0.0,
105
+ record: str | None = None,
106
+ ) -> dict:
107
+ """Render ``url`` in Chromium and save a screenshot."""
108
+ sync_playwright = _load_pw()
109
+ log = _PageLog()
110
+ with sync_playwright() as pw:
111
+ record_dir = None if record is None else str(Path(record).parent)
112
+ browser, context, page = _open(
113
+ pw, url, width=width, height=height, timeout=timeout, profile=profile,
114
+ log=log, headless=headless, record_dir=record_dir,
115
+ )
116
+ try:
117
+ if settle > 0:
118
+ page.wait_for_timeout(int(settle * 1000))
119
+ page.screenshot(path=output, full_page=full_page)
120
+ result = {
121
+ "path": output,
122
+ "url": page.url,
123
+ "title": page.title(),
124
+ "viewport": {"width": width, "height": height},
125
+ "full_page": full_page,
126
+ "headless": headless,
127
+ }
128
+ result.update(log.as_dict())
129
+ if hold > 0:
130
+ page.wait_for_timeout(int(hold * 1000))
131
+ return result
132
+ finally:
133
+ video = _close(browser, context, page=page, video_path=record)
134
+ if video:
135
+ result["video"] = video
136
+
137
+
138
+ def _apply_fills(page, fills: list[str], visible: bool = False) -> list[str]:
139
+ """Fill inputs as ``placeholder=text`` pairs; returns error strings.
140
+
141
+ With ``visible=True`` types character by character so a human watching a
142
+ headed browser can see the text appear.
143
+ """
144
+ errors = []
145
+ for item in fills:
146
+ if "=" not in item:
147
+ errors.append(f"invalid --fill {item!r}; expected placeholder=text")
148
+ continue
149
+ placeholder, text = item.split("=", 1)
150
+ locator = page.get_by_placeholder(placeholder, exact=True)
151
+ if locator.count() == 0:
152
+ locator = page.get_by_placeholder(placeholder)
153
+ if locator.count() != 1:
154
+ errors.append(
155
+ f"--fill {placeholder!r}: expected 1 input, got {locator.count()}"
156
+ )
157
+ continue
158
+ if visible:
159
+ locator.first.click()
160
+ locator.first.press_sequentially(text, delay=90)
161
+ else:
162
+ locator.first.fill(text)
163
+ return errors
164
+
165
+
166
+ def _flash_ring(page, cx: float, cy: float, radius: float) -> None:
167
+ """Draw a temporary orange halo at (cx, cy): transparent core, dense rim."""
168
+ page.evaluate(
169
+ """([x, y, r]) => {
170
+ const el = document.createElement('div');
171
+ el.style.cssText = 'position:fixed;pointer-events:none;z-index:2147483647;'
172
+ + 'left:' + (x - r) + 'px;top:' + (y - r) + 'px;'
173
+ + 'width:' + (2 * r) + 'px;height:' + (2 * r) + 'px;'
174
+ + 'border-radius:50%;'
175
+ + 'background:radial-gradient(circle,'
176
+ + ' rgba(255,140,0,0) 30%, rgba(255,140,0,0.85) 55%,'
177
+ + ' rgba(255,140,0,0.30) 75%, rgba(255,140,0,0) 100%);';
178
+ document.body.appendChild(el);
179
+ setTimeout(() => el.remove(), 1400);
180
+ }""",
181
+ [cx, cy, radius],
182
+ )
183
+
184
+
185
+ def click(
186
+ url: str,
187
+ target: str | None,
188
+ *,
189
+ selector: str | None = None,
190
+ contains: bool = False,
191
+ fills: list[str] | None = None,
192
+ width: int = 1280,
193
+ height: int = 800,
194
+ timeout: float = 15.0,
195
+ settle: float = 0.5,
196
+ profile: str | None = None,
197
+ headless: bool = True,
198
+ hold: float = 0.0,
199
+ expect: str | None = None,
200
+ expect_timeout: float = 10.0,
201
+ record: str | None = None,
202
+ before_path: str | None = None,
203
+ after_path: str | None = None,
204
+ ) -> dict:
205
+ """Open ``url``, optionally fill inputs, DOM-click a unique ``target``."""
206
+ sync_playwright = _load_pw()
207
+ log = _PageLog()
208
+ with sync_playwright() as pw:
209
+ record_dir = None if record is None else str(Path(record).parent)
210
+ browser, context, page = _open(
211
+ pw, url, width=width, height=height, timeout=timeout, profile=profile,
212
+ log=log, headless=headless, record_dir=record_dir,
213
+ )
214
+ try:
215
+ if before_path:
216
+ page.screenshot(path=before_path, full_page=False)
217
+
218
+ demo = not headless or record is not None
219
+ fill_errors = _apply_fills(page, fills or [], visible=demo)
220
+
221
+ if selector:
222
+ locator = page.locator(selector)
223
+ else:
224
+ locator = page.get_by_text(target, exact=True)
225
+ count = locator.count()
226
+ if count == 0 and not selector and contains:
227
+ locator = page.get_by_text(target)
228
+ count = locator.count()
229
+ candidates = []
230
+ for i in range(min(count, 8)):
231
+ item = locator.nth(i)
232
+ box = item.bounding_box()
233
+ candidates.append(
234
+ {
235
+ "text": (item.text_content() or "").strip(),
236
+ "bbox": (
237
+ None
238
+ if box is None
239
+ else [
240
+ box["x"],
241
+ box["y"],
242
+ box["x"] + box["width"],
243
+ box["y"] + box["height"],
244
+ ]
245
+ ),
246
+ }
247
+ )
248
+ result = {
249
+ "clicked": False,
250
+ "url": page.url,
251
+ "target": selector or target,
252
+ "candidate_count": count,
253
+ "candidates": candidates,
254
+ "fill_errors": fill_errors,
255
+ "before": before_path,
256
+ "after": None,
257
+ "x": None,
258
+ "y": None,
259
+ }
260
+ if count != 1:
261
+ result["error"] = "no match" if count == 0 else f"ambiguous: {count} matches"
262
+ result.update(log.as_dict())
263
+ return result
264
+ box = locator.first.bounding_box()
265
+ if demo and box is not None:
266
+ page.wait_for_timeout(400)
267
+ _flash_ring(
268
+ page,
269
+ box["x"] + box["width"] / 2,
270
+ box["y"] + box["height"] / 2,
271
+ max(14.0, min(32.0, float(min(box["width"], box["height"])))),
272
+ )
273
+ page.wait_for_timeout(900)
274
+ locator.first.click()
275
+ page.wait_for_timeout(int(settle * 1000))
276
+ verified = None
277
+ if expect:
278
+ try:
279
+ page.get_by_text(expect).first.wait_for(
280
+ state="visible", timeout=int(expect_timeout * 1000)
281
+ )
282
+ verified = True
283
+ except Exception:
284
+ verified = False
285
+ if after_path:
286
+ page.screenshot(path=after_path, full_page=False)
287
+ result.update(
288
+ {
289
+ "clicked": True,
290
+ "x": None if box is None else round(box["x"] + box["width"] / 2),
291
+ "y": None if box is None else round(box["y"] + box["height"] / 2),
292
+ "after": after_path,
293
+ "expect": expect,
294
+ "verified": verified,
295
+ }
296
+ )
297
+ result.update(log.as_dict())
298
+ if hold > 0:
299
+ page.wait_for_timeout(int(hold * 1000))
300
+ return result
301
+ finally:
302
+ video = _close(browser, context, page=page, video_path=record)
303
+ if video:
304
+ result["video"] = video
305
+
306
+
307
+ def snapshot(
308
+ url: str,
309
+ *,
310
+ width: int = 1280,
311
+ height: int = 800,
312
+ timeout: float = 15.0,
313
+ settle: float = 0.0,
314
+ profile: str | None = None,
315
+ headless: bool = True,
316
+ hold: float = 0.0,
317
+ ) -> dict:
318
+ """Return the page's accessibility tree as text (no vision model needed)."""
319
+ sync_playwright = _load_pw()
320
+ log = _PageLog()
321
+ with sync_playwright() as pw:
322
+ browser, context, page = _open(
323
+ pw, url, width=width, height=height, timeout=timeout, profile=profile,
324
+ log=log, headless=headless,
325
+ )
326
+ try:
327
+ if settle > 0:
328
+ page.wait_for_timeout(int(settle * 1000))
329
+ tree = page.locator("body").aria_snapshot()
330
+ result = {
331
+ "url": page.url,
332
+ "title": page.title(),
333
+ "snapshot": tree,
334
+ "headless": headless,
335
+ }
336
+ result.update(log.as_dict())
337
+ if hold > 0:
338
+ page.wait_for_timeout(int(hold * 1000))
339
+ return result
340
+ finally:
341
+ _close(browser, context)