browserwright 0.8.2__tar.gz → 0.9.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {browserwright-0.8.2/src/browserwright.egg-info → browserwright-0.9.0}/PKG-INFO +2 -1
- {browserwright-0.8.2 → browserwright-0.9.0}/pyproject.toml +26 -1
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/__init__.py +2 -1
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/cli.py +152 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/extension_upstream.py +12 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/errors.py +28 -0
- browserwright-0.9.0/src/browserwright/repl/_md_convert.py +101 -0
- browserwright-0.9.0/src/browserwright/repl/_md_normalize.py +234 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/repl/_namespace.py +9 -0
- browserwright-0.9.0/src/browserwright/repl/markdown.py +260 -0
- browserwright-0.9.0/src/browserwright/repl/vendor/README.md +44 -0
- browserwright-0.9.0/src/browserwright/repl/vendor/readability.js +2812 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/session_runtime.py +12 -3
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/skill_runtime.md +49 -8
- {browserwright-0.8.2 → browserwright-0.9.0/src/browserwright.egg-info}/PKG-INFO +2 -1
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright.egg-info/SOURCES.txt +5 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright.egg-info/requires.txt +1 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/LICENSE +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/README.md +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/setup.cfg +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/__main__.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/_executor/__init__.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/_executor/__main__.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/_executor/client.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/_executor/process.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/_executor/protocol.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/cdp.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/__init__.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/_ipc.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/_net.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/_rpc.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/_stale.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/backends/__init__.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/backends/base.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/backends/cdp.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/backends/extension.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/cli.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/config.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/doctor.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/errors.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/launch_chrome.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/launchagent.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/observability.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/platforms.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/probe.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/relay_status.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/resolver.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/__init__.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/daemon.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/executor_registry.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/facade.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/facade_extension.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/listener.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/proxy.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/relay.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/state.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/status.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/upstream.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/verbs.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/supervise.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/userscripts.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/discovery.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/health.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/install.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/memory/__init__.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/memory/_lock.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/memory/_md.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/memory/_yaml.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/memory/global_mem.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/memory/site_mem.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/mode_b_client.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/output_schema.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/primitives/__init__.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/primitives/discovery_api.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/primitives/http.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/primitives/site.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/repl/__init__.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/repl/_smart_goto.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/repl/inline.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/repl/playwright_handle.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/repl/snapshot.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/session.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/session_create.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/session_ctx.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/session_registry.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/github.com/SKILL.md +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/github.com/memory.md +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/github.com/tasks/list_issues.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/google.com/SKILL.md +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/google.com/memory.md +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/google.com/tasks/search.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/producthunt.com/SKILL.md +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/producthunt.com/memory.md +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/producthunt.com/tasks/today.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/wikipedia.org/SKILL.md +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/wikipedia.org/memory.md +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/wikipedia.org/tasks/lookup.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/ycombinator.com/SKILL.md +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/ycombinator.com/memory.md +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/ycombinator.com/tasks/front_page.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/skill_doc.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/task_runner.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/version.py +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright.egg-info/dependency_links.txt +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright.egg-info/entry_points.txt +0 -0
- {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: browserwright
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.9.0
|
|
4
4
|
Summary: Browserwright — let AI/code agents drive a real or isolated browser and author userscripts. Single package: the agent-facing REPL/site-skills/memory layer plus the bundled browser-resolving daemon (CDP proxy + extension/cloud backends).
|
|
5
5
|
License-Expression: AGPL-3.0-only
|
|
6
6
|
Requires-Python: >=3.11
|
|
@@ -10,6 +10,7 @@ Requires-Dist: pillow==12.2.0
|
|
|
10
10
|
Requires-Dist: httpx>=0.27
|
|
11
11
|
Requires-Dist: playwright>=1.60.0
|
|
12
12
|
Requires-Dist: pyyaml>=6.0.3
|
|
13
|
+
Requires-Dist: html-to-markdown<4,>=3.10.6
|
|
13
14
|
Provides-Extra: ux
|
|
14
15
|
Requires-Dist: rich>=13; extra == "ux"
|
|
15
16
|
Dynamic: license-file
|
|
@@ -15,7 +15,7 @@ name = "browserwright"
|
|
|
15
15
|
# stamping regex is `^(version\s*=\s*)["'][^"']+["']\s*$` and CI aborts unless
|
|
16
16
|
# it matches exactly once. `tests/skill/test_release_versioning.py` enforces all
|
|
17
17
|
# of that before a tag is ever pushed. See RELEASING.md.
|
|
18
|
-
version = "0.
|
|
18
|
+
version = "0.9.0"
|
|
19
19
|
description = "Browserwright — let AI/code agents drive a real or isolated browser and author userscripts. Single package: the agent-facing REPL/site-skills/memory layer plus the bundled browser-resolving daemon (CDP proxy + extension/cloud backends)."
|
|
20
20
|
requires-python = ">=3.11"
|
|
21
21
|
license = "AGPL-3.0-only"
|
|
@@ -30,6 +30,26 @@ dependencies = [
|
|
|
30
30
|
# daemon-resolved Chrome and does not need Playwright's bundled browser.
|
|
31
31
|
"playwright>=1.60.0",
|
|
32
32
|
"pyyaml>=6.0.3",
|
|
33
|
+
# HTML -> Markdown for the `read_markdown()` content view (ADR-0006/0007).
|
|
34
|
+
# Rust core + PyO3, and it pulls in NOTHING transitively — which is why it
|
|
35
|
+
# beat `markdownify` (bs4 + soupsieve) and `trafilatura` (lxml + a 32 MB
|
|
36
|
+
# babel locale tree) here. It is also the only converter measured that
|
|
37
|
+
# carries `class="language-*"` into a fenced ```lang block and keeps
|
|
38
|
+
# `rowspan` cells, both of which matter on the documentation pages this
|
|
39
|
+
# view exists to read.
|
|
40
|
+
#
|
|
41
|
+
# We deliberately track upstream rather than freeze: the project is running
|
|
42
|
+
# a systematic markdown-fidelity campaign (its own "Phase Z..HH" feat
|
|
43
|
+
# series) that keeps improving exactly what we depend on. The safety net for
|
|
44
|
+
# that is `tests/skill/test_markdown_golden.py` — golden files turn an
|
|
45
|
+
# upstream byte-level output change into a CI failure instead of a silent
|
|
46
|
+
# shift in what every agent reads. ADR-0007 makes those tests a precondition
|
|
47
|
+
# of this `>=`, not an optional extra.
|
|
48
|
+
#
|
|
49
|
+
# The `<4` cap is not conservatism about features: 3.0.0 removed the v1
|
|
50
|
+
# `convert_to_markdown()` entry point outright, so a major bump can break us
|
|
51
|
+
# at import time. Raise it deliberately, with the golden files as the diff.
|
|
52
|
+
"html-to-markdown>=3.10.6,<4",
|
|
33
53
|
]
|
|
34
54
|
|
|
35
55
|
[project.optional-dependencies]
|
|
@@ -58,6 +78,11 @@ browserwright = [
|
|
|
58
78
|
"site_skills_starter/**/*.md",
|
|
59
79
|
"site_skills_starter/**/*.py",
|
|
60
80
|
"skill_runtime.md",
|
|
81
|
+
# Vendored Mozilla Readability, injected as SOURCE TEXT via page.evaluate
|
|
82
|
+
# (add_script_tag is blocked by page CSP; page.evaluate is not). It has to
|
|
83
|
+
# ship as a real file for that reason — see repl/vendor/README.md.
|
|
84
|
+
"repl/vendor/*.js",
|
|
85
|
+
"repl/vendor/*.md",
|
|
61
86
|
]
|
|
62
87
|
|
|
63
88
|
[tool.setuptools.exclude-package-data]
|
|
@@ -42,6 +42,7 @@ from .errors import ( # noqa: F401
|
|
|
42
42
|
NeedsUserConfirm,
|
|
43
43
|
NetworkError,
|
|
44
44
|
PageLoadFailed,
|
|
45
|
+
UnsupportedContentType,
|
|
45
46
|
)
|
|
46
47
|
from .primitives.discovery_api import ( # noqa: F401
|
|
47
48
|
list_site_skills,
|
|
@@ -69,7 +70,7 @@ EXPORTS = [
|
|
|
69
70
|
# errors
|
|
70
71
|
"BrowserwrightError", "PageLoadFailed", "ElementNotFound", "AuthWall",
|
|
71
72
|
"Captcha", "NetworkError", "DaemonUnavailable", "CDPError",
|
|
72
|
-
"NeedsUserConfirm",
|
|
73
|
+
"NeedsUserConfirm", "UnsupportedContentType",
|
|
73
74
|
]
|
|
74
75
|
|
|
75
76
|
__all__ = EXPORTS
|
|
@@ -38,6 +38,11 @@ Usage:
|
|
|
38
38
|
browserwright whoami --session=ID
|
|
39
39
|
browserwright userscript {push|list|remove|toggle|logs} ...
|
|
40
40
|
|
|
41
|
+
browserwright markdown <url> [--mode=auto|article|full] [--backend=extension|cdp]
|
|
42
|
+
[--out=PATH] [--max-chars=N]
|
|
43
|
+
One page as Markdown. Creates and tears down its own session, so it takes
|
|
44
|
+
no -s. Absolute links, shadow DOM flattened in, HTML only.
|
|
45
|
+
|
|
41
46
|
browserwright -s <session-id> task <site>/<name> [--key=value ...] [--isolated]
|
|
42
47
|
browserwright list-tasks [--site SITE] [--query Q] [--json]
|
|
43
48
|
|
|
@@ -418,6 +423,149 @@ def _cmd_task(args: list[str], *, session_id: Optional[str] = None) -> int:
|
|
|
418
423
|
return 0
|
|
419
424
|
|
|
420
425
|
|
|
426
|
+
MARKDOWN_HELP = """Usage:
|
|
427
|
+
browserwright markdown <url> [--mode=auto|article|full] [--backend=extension|cdp]
|
|
428
|
+
[--out=PATH] [--max-chars=N] [--name=LABEL]
|
|
429
|
+
|
|
430
|
+
Fetch one page and print it as Markdown. Owns its whole lifecycle: it creates a
|
|
431
|
+
throwaway session, navigates, converts, and tears the session down again — so
|
|
432
|
+
unlike every other browser-driving command it takes no -s/--session.
|
|
433
|
+
|
|
434
|
+
Links come out absolute, open shadow roots are flattened in, and same-origin
|
|
435
|
+
iframes are inlined. Only HTML is converted; anything else is refused with its
|
|
436
|
+
Content-Type rather than silently returning an empty-looking result.
|
|
437
|
+
|
|
438
|
+
Flags:
|
|
439
|
+
--mode=auto (default) extract the main content; if extraction collapses,
|
|
440
|
+
fall back to the page with nav/aside/footer/forms removed
|
|
441
|
+
--mode=article force extraction
|
|
442
|
+
--mode=full the page verbatim — use it when you want the navigation,
|
|
443
|
+
a form, or every link
|
|
444
|
+
--backend extension (default, the user's real Chrome) | cdp
|
|
445
|
+
--out=PATH write the full Markdown here instead of a temp file
|
|
446
|
+
--max-chars=N cap what is printed (default 8000; 0 prints everything).
|
|
447
|
+
The FULL text is always written to the file either way.
|
|
448
|
+
--name=LABEL session label; shows as the Chrome tab group title
|
|
449
|
+
"""
|
|
450
|
+
|
|
451
|
+
# Runs inside the session's executor, which is the only place a live `page`
|
|
452
|
+
# exists. Values arrive through the request environment (names only cross the
|
|
453
|
+
# CLI boundary, never values — same discipline as `--env`), and the markdown
|
|
454
|
+
# rides back on disk rather than through the executor's 10000-char console cap.
|
|
455
|
+
_MARKDOWN_CODE = """
|
|
456
|
+
import os
|
|
457
|
+
from browserwright.repl.markdown import render_page_markdown
|
|
458
|
+
|
|
459
|
+
page.goto(os.environ["BW_MD_URL"])
|
|
460
|
+
_r = render_page_markdown(page, mode=os.environ["BW_MD_MODE"])
|
|
461
|
+
with open(os.environ["BW_MD_OUT"], "w", encoding="utf-8") as _f:
|
|
462
|
+
_f.write(_r.markdown)
|
|
463
|
+
_bw_warn("markdown: rendered as %s (%d chars)" % (_r.mode_used, len(_r.markdown)))
|
|
464
|
+
for _n in _r.notes:
|
|
465
|
+
_bw_warn("markdown: " + _n)
|
|
466
|
+
"""
|
|
467
|
+
|
|
468
|
+
|
|
469
|
+
def _cmd_markdown(args: list[str]) -> int:
|
|
470
|
+
"""``browserwright markdown <url>`` — one page, as Markdown (ADR-0006).
|
|
471
|
+
|
|
472
|
+
The one browser-driving command with no session argument: it mints a
|
|
473
|
+
throwaway one, uses it, and ends it. Teardown runs in a `finally` because a
|
|
474
|
+
leaked session leaves a tab group behind in the user's real Chrome (see
|
|
475
|
+
issue #53 for why that path is worth watching).
|
|
476
|
+
"""
|
|
477
|
+
if not args or args[0] in {"-h", "--help"}:
|
|
478
|
+
sys.stdout.write(MARKDOWN_HELP)
|
|
479
|
+
return 0 if args else 1
|
|
480
|
+
|
|
481
|
+
url = args[0]
|
|
482
|
+
if url.startswith("-"):
|
|
483
|
+
print(f"usage error: expected a URL, got {url!r}", file=sys.stderr)
|
|
484
|
+
print(MARKDOWN_HELP, file=sys.stderr)
|
|
485
|
+
return 1
|
|
486
|
+
kw = _parse_kv_args(args[1:])
|
|
487
|
+
|
|
488
|
+
mode = str(kw.get("mode", "auto"))
|
|
489
|
+
from .repl.markdown import DEFAULT_MAX_CHARS, MODES, spill_path
|
|
490
|
+
|
|
491
|
+
if mode not in MODES:
|
|
492
|
+
print(f"usage error: --mode must be one of {'|'.join(MODES)}, "
|
|
493
|
+
f"got {mode!r}", file=sys.stderr)
|
|
494
|
+
return 1
|
|
495
|
+
backend = str(kw.get("backend", "extension"))
|
|
496
|
+
if backend not in ("extension", "cdp"):
|
|
497
|
+
print(f"usage error: --backend must be extension or cdp, got "
|
|
498
|
+
f"{backend!r}", file=sys.stderr)
|
|
499
|
+
return 1
|
|
500
|
+
try:
|
|
501
|
+
max_chars = int(kw.get("max-chars", DEFAULT_MAX_CHARS))
|
|
502
|
+
except (TypeError, ValueError):
|
|
503
|
+
print("usage error: --max-chars must be an integer (0 = no cap)",
|
|
504
|
+
file=sys.stderr)
|
|
505
|
+
return 1
|
|
506
|
+
|
|
507
|
+
out = kw.get("out")
|
|
508
|
+
keep_file = out is not None
|
|
509
|
+
out_path = str(out) if keep_file else spill_path(url)
|
|
510
|
+
|
|
511
|
+
import contextlib
|
|
512
|
+
|
|
513
|
+
from . import session_create
|
|
514
|
+
from .errors import BrowserwrightError
|
|
515
|
+
from .repl import inline
|
|
516
|
+
from .session_ctx import resolve_session_or_env
|
|
517
|
+
|
|
518
|
+
try:
|
|
519
|
+
sid = session_create.new(
|
|
520
|
+
backend=backend,
|
|
521
|
+
# `cdp` has no meaning without an owner: it must either launch a
|
|
522
|
+
# browser or attach to one, and a throwaway session has nobody to
|
|
523
|
+
# attach to. NOTE this launches a real Chrome, which on macOS takes
|
|
524
|
+
# the active window — the default backend is `extension` precisely
|
|
525
|
+
# so the common path never does that.
|
|
526
|
+
create=(backend == "cdp"),
|
|
527
|
+
attach=None,
|
|
528
|
+
name=str(kw.get("name", "markdown")),
|
|
529
|
+
)
|
|
530
|
+
except (ValueError, BrowserwrightError) as e:
|
|
531
|
+
print(str(e), file=sys.stderr)
|
|
532
|
+
return 1
|
|
533
|
+
|
|
534
|
+
try:
|
|
535
|
+
rc = inline.run_code(
|
|
536
|
+
_MARKDOWN_CODE,
|
|
537
|
+
session_id=sid,
|
|
538
|
+
env={"BW_MD_URL": url, "BW_MD_OUT": out_path, "BW_MD_MODE": mode},
|
|
539
|
+
)
|
|
540
|
+
if rc != 0:
|
|
541
|
+
return rc
|
|
542
|
+
try:
|
|
543
|
+
text = Path(out_path).read_text(encoding="utf-8")
|
|
544
|
+
except OSError as e:
|
|
545
|
+
print(f"markdown: could not read rendered output: {e}",
|
|
546
|
+
file=sys.stderr)
|
|
547
|
+
return 3
|
|
548
|
+
if max_chars > 0 and len(text) > max_chars:
|
|
549
|
+
from .repl.snapshot import _truncate_lines
|
|
550
|
+
|
|
551
|
+
print(f"[markdown] truncated to {max_chars} of {len(text)} chars; "
|
|
552
|
+
f"full text: {out_path}", file=sys.stderr)
|
|
553
|
+
sys.stdout.write(_truncate_lines(text, max_chars) + "\n")
|
|
554
|
+
keep_file = True # the caller needs it to see the rest
|
|
555
|
+
else:
|
|
556
|
+
sys.stdout.write(text if text.endswith("\n") else text + "\n")
|
|
557
|
+
return 0
|
|
558
|
+
finally:
|
|
559
|
+
if not keep_file:
|
|
560
|
+
with contextlib.suppress(OSError):
|
|
561
|
+
Path(out_path).unlink()
|
|
562
|
+
try:
|
|
563
|
+
session_create.end(resolve_session_or_env(sid))
|
|
564
|
+
except Exception as e: # noqa: BLE001 — never mask the real result
|
|
565
|
+
print(f"[markdown] warning: could not end throwaway session {sid}: "
|
|
566
|
+
f"{e}", file=sys.stderr)
|
|
567
|
+
|
|
568
|
+
|
|
421
569
|
def _cmd_doctor(args: list[str]) -> int:
|
|
422
570
|
"""A4: a ``{status, message, fix}`` check table.
|
|
423
571
|
|
|
@@ -890,6 +1038,10 @@ def main(argv: Optional[list[str]] = None) -> None:
|
|
|
890
1038
|
sys.exit(_cmd_index(rest))
|
|
891
1039
|
if cmd == "memory":
|
|
892
1040
|
sys.exit(_cmd_memory(rest))
|
|
1041
|
+
# Deliberately NOT under the -s branch: this is the one browser-driving
|
|
1042
|
+
# command that owns its own throwaway session (ADR-0006).
|
|
1043
|
+
if cmd == "markdown":
|
|
1044
|
+
sys.exit(_cmd_markdown(rest))
|
|
893
1045
|
if cmd == "session":
|
|
894
1046
|
sys.exit(_cmd_session(rest, session_id=global_session))
|
|
895
1047
|
if cmd == "whoami":
|
{browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/extension_upstream.py
RENAMED
|
@@ -1242,6 +1242,18 @@ class ExtensionUpstream:
|
|
|
1242
1242
|
group_id: int) -> dict:
|
|
1243
1243
|
resolved_group_id, info = await self._resolve_session_group(
|
|
1244
1244
|
session_id, group_id)
|
|
1245
|
+
# The adapter-memory fast path returns ``(gid, None)`` — the group
|
|
1246
|
+
# query is deferred to the caller. Mirror `_group_member_tabs` and
|
|
1247
|
+
# re-query by the resolved id before concluding the group is empty
|
|
1248
|
+
# (e2e recovery test: an in-daemon session that just opened a tab
|
|
1249
|
+
# hits the memory path and would otherwise be reported "group
|
|
1250
|
+
# missing" despite a live group with tabs).
|
|
1251
|
+
if info is None and isinstance(resolved_group_id, int) and resolved_group_id >= 0:
|
|
1252
|
+
query_generation = self._relay_generation()
|
|
1253
|
+
info = await self._relay.query_group_tabs(group_id=resolved_group_id)
|
|
1254
|
+
if query_generation != self._relay_generation():
|
|
1255
|
+
raise RuntimeError(
|
|
1256
|
+
"extension reconnected while resolving group membership")
|
|
1245
1257
|
if not info or not info.get("tabs"):
|
|
1246
1258
|
raise RuntimeError(
|
|
1247
1259
|
f"no recoverable tabs for group id {group_id} "
|
|
@@ -75,6 +75,34 @@ class ElementNotFound(BrowserwrightError):
|
|
|
75
75
|
super().__init__(f"element not found: {selector!r} after {timeout}s", fix=fix)
|
|
76
76
|
|
|
77
77
|
|
|
78
|
+
class UnsupportedContentType(BrowserwrightError):
|
|
79
|
+
"""The page is not HTML, so there is nothing to convert to Markdown.
|
|
80
|
+
|
|
81
|
+
Raised loudly on purpose (ADR-0007). A PDF is the motivating case: Chrome's
|
|
82
|
+
built-in viewer renders a REAL DOM (an ``<embed>`` shell), so a best-effort
|
|
83
|
+
conversion would succeed and return a short, plausible-looking result that
|
|
84
|
+
contains none of the document — a silent empty answer, which is the one
|
|
85
|
+
failure mode the markdown path is built to make impossible.
|
|
86
|
+
|
|
87
|
+
browserwright deliberately does not grow a document pipeline. Route the URL
|
|
88
|
+
to whatever handles that type and keep browserwright for HTML.
|
|
89
|
+
"""
|
|
90
|
+
|
|
91
|
+
exit_code = 3
|
|
92
|
+
default_fix = (
|
|
93
|
+
"this endpoint only converts HTML; route non-HTML types "
|
|
94
|
+
"(PDF, images, binaries) to a handler for that type"
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
def __init__(self, url: str = "", content_type: str = "", fix: str = ""):
|
|
98
|
+
self.url, self.content_type = url, content_type
|
|
99
|
+
super().__init__(
|
|
100
|
+
f"not HTML, refusing to convert: {url} (Content-Type: "
|
|
101
|
+
f"{content_type or 'unknown'})",
|
|
102
|
+
fix=fix,
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
|
|
78
106
|
class AuthWall(BrowserwrightError):
|
|
79
107
|
exit_code = 4
|
|
80
108
|
default_fix = "stop and ask the user to log in; do not type credentials from a screenshot"
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""The one and only place HTML becomes Markdown.
|
|
2
|
+
|
|
3
|
+
ADR-0007 requires this seam: we track `html-to-markdown` with a `>=` rather
|
|
4
|
+
than freezing it, so the day we swap converters — or the day upstream changes a
|
|
5
|
+
default out from under us — exactly one function has to change, and
|
|
6
|
+
``tests/skill/test_markdown_golden.py`` is what makes the change visible.
|
|
7
|
+
|
|
8
|
+
Nothing else in the tree may import ``html_to_markdown`` directly.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
# Built once per variant. `ConversionOptions` is a plain frozen options object,
|
|
15
|
+
# so module singletons are safe and save rebuilding 44 fields per page.
|
|
16
|
+
_OPTIONS: dict[bool, Any] = {}
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _options(strip_chrome: bool) -> Any:
|
|
20
|
+
if strip_chrome not in _OPTIONS:
|
|
21
|
+
from html_to_markdown import ConversionOptions, PreprocessingOptions
|
|
22
|
+
|
|
23
|
+
# Two tiers of "remove things", and the difference is the whole point:
|
|
24
|
+
#
|
|
25
|
+
# strip_chrome=False -> remove NOTHING. The verbatim page.
|
|
26
|
+
# strip_chrome=True -> remove KNOWN chrome by tag/class (`<nav>`,
|
|
27
|
+
# `<aside>`, `<footer>`, forms, `.sidebar`…).
|
|
28
|
+
#
|
|
29
|
+
# Neither one scores or guesses which block is the article — that is
|
|
30
|
+
# Readability's job, upstream of here, and the reason this second tier
|
|
31
|
+
# exists at all is that Readability's guessing can wipe a page out
|
|
32
|
+
# (a GitHub issue page: 118 links -> 0, 2866 tokens -> 92). Tag-based
|
|
33
|
+
# removal cannot do that: it deletes the elements it recognizes and
|
|
34
|
+
# keeps everything else, so it is the honest fallback when extraction
|
|
35
|
+
# collapses — the reader wanted the body text, not the site furniture.
|
|
36
|
+
#
|
|
37
|
+
# Measured on 3.10.6 with nav/header/main/aside/footer/.sidebar/form:
|
|
38
|
+
# enabled=False keeps all 8 markers
|
|
39
|
+
# preset="standard" drops only <nav> and <form>
|
|
40
|
+
# preset="aggressive" keeps header + body text + body links; drops the
|
|
41
|
+
# rest of the furniture
|
|
42
|
+
pre = (
|
|
43
|
+
PreprocessingOptions(enabled=True, preset="aggressive")
|
|
44
|
+
if strip_chrome
|
|
45
|
+
else PreprocessingOptions(enabled=False)
|
|
46
|
+
)
|
|
47
|
+
_OPTIONS[strip_chrome] = ConversionOptions(
|
|
48
|
+
# Never left at its default. `preprocessing=None` is NOT "off" — it
|
|
49
|
+
# means "use the built-in default", which is
|
|
50
|
+
# `enabled=True, remove_navigation=True, remove_forms=True`:
|
|
51
|
+
# '<nav><a href="/n">nav link</a></nav><p>body</p><form>…</form>'
|
|
52
|
+
# default -> 'body\n'
|
|
53
|
+
# enabled=False -> '[nav link](/n)\n\nbody\n\nQ\n'
|
|
54
|
+
# So out of the box this library quietly deletes navigation and
|
|
55
|
+
# forms on EVERY conversion, including the one the caller asked to
|
|
56
|
+
# be verbatim. Removal has to be our decision, driven by `mode`, and
|
|
57
|
+
# announced — never a converter default nobody chose.
|
|
58
|
+
preprocessing=pre,
|
|
59
|
+
# Default is False, and leaving it there pads every separator row
|
|
60
|
+
# out to the widest cell in the column: a wide Wikipedia table
|
|
61
|
+
# measured 984,849 chars vs 359,598 with this on. That single flag
|
|
62
|
+
# is the whole of this library's "token-hungry" reputation.
|
|
63
|
+
compact_tables=True,
|
|
64
|
+
# Default is True, which prepends a YAML front-matter block built
|
|
65
|
+
# from <title>/<meta> whenever the document has a <head>. The agent
|
|
66
|
+
# asked for page content, not for a document envelope, and every
|
|
67
|
+
# downstream consumer would have to strip it (ADR-0007: metadata
|
|
68
|
+
# goes out-of-band, never into the body).
|
|
69
|
+
extract_metadata=False,
|
|
70
|
+
# Already the upstream default today. Pinned explicitly anyway
|
|
71
|
+
# because we track upstream with `>=`: ATX (`# h1`) vs setext
|
|
72
|
+
# (`h1\n===`) changes the bytes of every heading, and a default
|
|
73
|
+
# flip should break a golden file, not silently reshape output.
|
|
74
|
+
heading_style="atx",
|
|
75
|
+
)
|
|
76
|
+
return _OPTIONS[strip_chrome]
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def convert_html(html: str, *, strip_chrome: bool = False) -> str:
|
|
80
|
+
"""Convert an HTML string to Markdown. Returns "" for empty input.
|
|
81
|
+
|
|
82
|
+
``strip_chrome=True`` additionally removes recognized page furniture
|
|
83
|
+
(nav / aside / footer / forms / sidebar-ish class names) by tag and class,
|
|
84
|
+
never by scoring. Use it for the fallback path where the reader wants body
|
|
85
|
+
text; leave it off wherever the caller asked for the page verbatim, and off
|
|
86
|
+
for a Readability fragment, which has already been stripped.
|
|
87
|
+
|
|
88
|
+
The caller owns the empty-output decision. This function does NOT guard
|
|
89
|
+
against the upstream defect where a stray ``<td>``/``<th>`` outside a table
|
|
90
|
+
silently collapses its whole subtree to ``""`` (verified against 3.10.6:
|
|
91
|
+
``<div><td>hello</td></div>`` -> ``""``). That defect cannot fire on
|
|
92
|
+
browser-serialized HTML, which is always well-formed — only on hand-built
|
|
93
|
+
fragments, i.e. the Readability path — so the guard lives where the
|
|
94
|
+
fragment is produced, as the "empty -> fall back to the full page" rule of
|
|
95
|
+
ADR-0007.
|
|
96
|
+
"""
|
|
97
|
+
if not html or not html.strip():
|
|
98
|
+
return ""
|
|
99
|
+
from html_to_markdown import convert
|
|
100
|
+
|
|
101
|
+
return convert(html, _options(strip_chrome)).content
|
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
"""The in-page half of the markdown pipeline: normalize the live DOM to HTML.
|
|
2
|
+
|
|
3
|
+
This exists because ``page.content()`` silently loses two things Markdown needs,
|
|
4
|
+
and neither can be recovered on the Python side (ADR-0007):
|
|
5
|
+
|
|
6
|
+
- **open shadow roots are not serialized at all.** Verified in this repo: a page
|
|
7
|
+
with a declarative ``<template shadowrootmode="open">`` reports ``shadowRoot``
|
|
8
|
+
present, in-page JS reads its text fine, and the string from
|
|
9
|
+
``page.content()`` does not contain it.
|
|
10
|
+
- **relative URLs stay relative.** No Python converter resolves ``<base>``. The
|
|
11
|
+
browser resolves per document — including separately inside each iframe.
|
|
12
|
+
|
|
13
|
+
So this script runs where the DOM is, rebuilds a detached copy with those two
|
|
14
|
+
problems fixed, and hands back plain HTML strings.
|
|
15
|
+
|
|
16
|
+
Why ``page.evaluate`` and not ``add_script_tag``
|
|
17
|
+
------------------------------------------------
|
|
18
|
+
Verified in this repo against ``default-src 'none'; script-src 'none'`` with
|
|
19
|
+
``bypass_csp=False``: the page's own inline ``<script>`` does NOT run (CSP is
|
|
20
|
+
being enforced), while ``page.evaluate("new Function('return 41+1')()")``
|
|
21
|
+
returns ``42`` and ``add_script_tag`` raises. CDP's ``Runtime.evaluate`` carries
|
|
22
|
+
``allowUnsafeEvalBlockedByCSP``, so protocol-driven evaluation is exempt where
|
|
23
|
+
page-driven script injection is not.
|
|
24
|
+
|
|
25
|
+
Three consequences the script below obeys:
|
|
26
|
+
|
|
27
|
+
- **Stay synchronous.** The CSP exemption is a toggle held for the duration of
|
|
28
|
+
the protocol call; a continuation scheduled from inside it (``setTimeout``,
|
|
29
|
+
``await``) loses it. Everything here is one synchronous pass.
|
|
30
|
+
- **Never assign ``innerHTML``, never use ``DOMParser``.** Those are gated by
|
|
31
|
+
Trusted Types, which ``page.evaluate`` does NOT bypass — it runs in the main
|
|
32
|
+
world. *Reading* ``innerHTML``/``outerHTML`` and calling ``cloneNode`` /
|
|
33
|
+
``importNode`` are unaffected, which is why the copy is built with node APIs
|
|
34
|
+
and HTML is only ever read out.
|
|
35
|
+
- **Never mutate the live page.** Everything is built detached, and Readability
|
|
36
|
+
(which rewrites whatever document it is handed) only ever sees a scratch
|
|
37
|
+
document.
|
|
38
|
+
|
|
39
|
+
And do not "fix" any of this by turning on ``bypass_csp``: that lets the page's
|
|
40
|
+
previously-blocked inline scripts run and stops Trusted Types being enforced,
|
|
41
|
+
i.e. it changes the very DOM we are here to capture.
|
|
42
|
+
"""
|
|
43
|
+
from __future__ import annotations
|
|
44
|
+
|
|
45
|
+
from functools import lru_cache
|
|
46
|
+
from pathlib import Path
|
|
47
|
+
|
|
48
|
+
_VENDOR = Path(__file__).resolve().parent / "vendor"
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@lru_cache(maxsize=1)
|
|
52
|
+
def _readability_source() -> str:
|
|
53
|
+
"""Mozilla Readability 0.6.0, vendored. See ``vendor/README.md``."""
|
|
54
|
+
return (_VENDOR / "readability.js").read_text(encoding="utf-8")
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
# The body of the injected function. Assembled with the Readability source in
|
|
58
|
+
# `build_script()` below, because the extraction pass needs it in scope.
|
|
59
|
+
_NORMALIZE_BODY = r"""
|
|
60
|
+
// Elements dropped wholesale: no readable content, pure token cost.
|
|
61
|
+
// `svg`/`canvas` are decorative here - when they carry an accessible name it
|
|
62
|
+
// lives on an ancestor, which survives. `template` is author-declared inert
|
|
63
|
+
// content; declarative shadow DOM is consumed by the parser long before this
|
|
64
|
+
// runs and surfaces as a real shadowRoot, handled below.
|
|
65
|
+
const DROP = new Set([
|
|
66
|
+
"SCRIPT", "STYLE", "NOSCRIPT", "TEMPLATE", "SVG", "CANVAS",
|
|
67
|
+
"LINK", "META", "BASE",
|
|
68
|
+
]);
|
|
69
|
+
// Attributes whose value is a URL we want absolute in the output.
|
|
70
|
+
const URL_ATTRS = { A: "href", AREA: "href", IMG: "src", SOURCE: "src" };
|
|
71
|
+
const MAX_DEPTH = 128;
|
|
72
|
+
|
|
73
|
+
const stats = {
|
|
74
|
+
shadowRoots: 0,
|
|
75
|
+
sameOriginFrames: 0,
|
|
76
|
+
crossOriginFrames: 0,
|
|
77
|
+
linksTotal: 0,
|
|
78
|
+
};
|
|
79
|
+
|
|
80
|
+
const isHidden = (el) => {
|
|
81
|
+
if (el.hasAttribute("hidden")) return true;
|
|
82
|
+
if (el.getAttribute("aria-hidden") === "true") return true;
|
|
83
|
+
return false;
|
|
84
|
+
};
|
|
85
|
+
|
|
86
|
+
// Resolve against the *owning* document's base URI, which differs per frame.
|
|
87
|
+
// Reading the IDL property (`a.href`) would resolve nodes imported from an
|
|
88
|
+
// iframe against the top document instead, so resolve explicitly.
|
|
89
|
+
const absolutize = (src, clone, baseURI) => {
|
|
90
|
+
const attr = URL_ATTRS[src.nodeName.toUpperCase()];
|
|
91
|
+
if (!attr) return;
|
|
92
|
+
const raw = src.getAttribute(attr);
|
|
93
|
+
if (!raw) return;
|
|
94
|
+
if (attr === "href") stats.linksTotal += 1;
|
|
95
|
+
try {
|
|
96
|
+
clone.setAttribute(attr, new URL(raw, baseURI).href);
|
|
97
|
+
} catch (e) {
|
|
98
|
+
/* mailto:, javascript:, malformed - keep the author's value */
|
|
99
|
+
}
|
|
100
|
+
};
|
|
101
|
+
|
|
102
|
+
// Build a detached copy of `node`. Returns null when the node is dropped.
|
|
103
|
+
const build = (node, baseURI, depth) => {
|
|
104
|
+
if (depth > MAX_DEPTH) return null;
|
|
105
|
+
|
|
106
|
+
if (node.nodeType === Node.TEXT_NODE) {
|
|
107
|
+
return node.data.trim() ? node.cloneNode(false) : null;
|
|
108
|
+
}
|
|
109
|
+
if (node.nodeType !== Node.ELEMENT_NODE) return null;
|
|
110
|
+
|
|
111
|
+
const name = node.nodeName.toUpperCase();
|
|
112
|
+
if (DROP.has(name)) return null;
|
|
113
|
+
if (isHidden(node)) return null;
|
|
114
|
+
|
|
115
|
+
if (name === "IFRAME") {
|
|
116
|
+
// A null contentDocument IS the same-origin test. Sandboxed frames can
|
|
117
|
+
// throw rather than return null, hence the guard.
|
|
118
|
+
let doc = null;
|
|
119
|
+
try { doc = node.contentDocument; } catch (e) { doc = null; }
|
|
120
|
+
if (!doc || !doc.body) {
|
|
121
|
+
stats.crossOriginFrames += 1;
|
|
122
|
+
return null;
|
|
123
|
+
}
|
|
124
|
+
stats.sameOriginFrames += 1;
|
|
125
|
+
// Inline the frame body at the iframe's position, wrapped so the
|
|
126
|
+
// converter sees a block boundary instead of splicing the frame's
|
|
127
|
+
// paragraphs into the parent's.
|
|
128
|
+
const holder = document.createElement("div");
|
|
129
|
+
for (const child of Array.from(doc.body.childNodes)) {
|
|
130
|
+
const built = build(child, doc.baseURI, depth + 1);
|
|
131
|
+
if (built) holder.appendChild(built);
|
|
132
|
+
}
|
|
133
|
+
return holder.childNodes.length ? holder : null;
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
// Shallow clone keeps attributes; children are rebuilt so we control
|
|
137
|
+
// shadow/slot/iframe expansion.
|
|
138
|
+
const clone = node.cloneNode(false);
|
|
139
|
+
absolutize(node, clone, baseURI);
|
|
140
|
+
|
|
141
|
+
// A SLOT renders its assigned nodes. Expanding it here is what makes a
|
|
142
|
+
// flattened shadow tree come out in composed-tree order rather than
|
|
143
|
+
// shadow-then-light.
|
|
144
|
+
if (name === "SLOT" && typeof node.assignedNodes === "function") {
|
|
145
|
+
const holder = document.createElement("div");
|
|
146
|
+
for (const child of node.assignedNodes({ flatten: true })) {
|
|
147
|
+
const built = build(child, baseURI, depth + 1);
|
|
148
|
+
if (built) holder.appendChild(built);
|
|
149
|
+
}
|
|
150
|
+
return holder.childNodes.length ? holder : null;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
// Open shadow root: its content is what the user actually sees, and the
|
|
154
|
+
// light children are reached through the slots inside it. A CLOSED root is
|
|
155
|
+
// indistinguishable from no root here - `shadowRoot` is null either way -
|
|
156
|
+
// so closed roots are invisible to every approach, ours included, and we
|
|
157
|
+
// cannot even report their number. See ADR-0007.
|
|
158
|
+
const root = node.shadowRoot;
|
|
159
|
+
if (root) {
|
|
160
|
+
stats.shadowRoots += 1;
|
|
161
|
+
for (const child of Array.from(root.childNodes)) {
|
|
162
|
+
const built = build(child, baseURI, depth + 1);
|
|
163
|
+
if (built) clone.appendChild(built);
|
|
164
|
+
}
|
|
165
|
+
return clone;
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
for (const child of Array.from(node.childNodes)) {
|
|
169
|
+
const built = build(child, baseURI, depth + 1);
|
|
170
|
+
if (built) clone.appendChild(built);
|
|
171
|
+
}
|
|
172
|
+
return clone;
|
|
173
|
+
};
|
|
174
|
+
|
|
175
|
+
// A scratch document, so Readability has something it is allowed to destroy
|
|
176
|
+
// and so the normalized tree is a real document rather than a loose node.
|
|
177
|
+
const scratch = document.implementation.createHTMLDocument(document.title || "");
|
|
178
|
+
const normalized = document.body ? build(document.body, document.baseURI, 0) : null;
|
|
179
|
+
if (normalized) {
|
|
180
|
+
for (const child of Array.from(normalized.childNodes)) {
|
|
181
|
+
scratch.body.appendChild(scratch.importNode(child, true));
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
const result = {
|
|
186
|
+
fullHtml: scratch.body.innerHTML,
|
|
187
|
+
articleHtml: null,
|
|
188
|
+
articleLinks: 0,
|
|
189
|
+
title: document.title || "",
|
|
190
|
+
url: location.href,
|
|
191
|
+
contentType: document.contentType || "",
|
|
192
|
+
stats: stats,
|
|
193
|
+
};
|
|
194
|
+
|
|
195
|
+
if (opts && opts.extract) {
|
|
196
|
+
// Readability rewrites the document it is given, so hand it a copy of the
|
|
197
|
+
// scratch document - never `scratch` itself, which we still need above.
|
|
198
|
+
try {
|
|
199
|
+
const forReader = scratch.cloneNode(true);
|
|
200
|
+
const article = new Readability(forReader).parse();
|
|
201
|
+
if (article && article.content) {
|
|
202
|
+
result.articleHtml = article.content;
|
|
203
|
+
// Counted here rather than in Python: the caller's collapse test is
|
|
204
|
+
// about the DOM that came out, and re-parsing markdown to count links
|
|
205
|
+
// would measure the converter instead.
|
|
206
|
+
const probe = document.implementation.createHTMLDocument("");
|
|
207
|
+
probe.body.appendChild(probe.importNode(forReader.body, true));
|
|
208
|
+
result.articleLinks = probe.querySelectorAll("a[href]").length;
|
|
209
|
+
}
|
|
210
|
+
} catch (e) {
|
|
211
|
+
result.articleHtml = null; // treated as a collapse by the caller
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
return result;
|
|
216
|
+
"""
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def build_script(*, extract: bool) -> str:
|
|
220
|
+
"""Assemble the full ``page.evaluate`` payload.
|
|
221
|
+
|
|
222
|
+
Returns the source of a single arrow function taking one options object.
|
|
223
|
+
The Readability source is inlined only when extraction is requested — it is
|
|
224
|
+
~91 KB, and shipping it across the CDP boundary on every full-page read
|
|
225
|
+
would be pure overhead.
|
|
226
|
+
"""
|
|
227
|
+
prelude = ""
|
|
228
|
+
if extract:
|
|
229
|
+
# Readability ends with `if (typeof module === "object") module.exports
|
|
230
|
+
# = Readability;`. In a page that leaked a bundler's `module` global
|
|
231
|
+
# (webpack/browserify do), that assignment would mutate the page. A
|
|
232
|
+
# local binding shadows it so the branch is never taken.
|
|
233
|
+
prelude = " var module = void 0;\n" + _readability_source() + "\n"
|
|
234
|
+
return "(opts) => {\n" + prelude + _NORMALIZE_BODY + "\n}"
|
|
@@ -101,6 +101,15 @@ def build_globals() -> dict[str, Any]:
|
|
|
101
101
|
# to override.
|
|
102
102
|
from .snapshot import make_snapshot
|
|
103
103
|
g["snapshot"] = make_snapshot(handle)
|
|
104
|
+
# The second view (ADR-0006): `snapshot()` is the action view, this is the
|
|
105
|
+
# content view. Same rules — injected per heredoc, read-only, never in
|
|
106
|
+
# EXPORTS (a view needs the live `page`, which a module-level function
|
|
107
|
+
# cannot have). `g` is passed so the closure can find the executor's
|
|
108
|
+
# `_bw_warn` channel, which is injected into this same dict *after* we
|
|
109
|
+
# return; its notes (extraction path taken, iframes excluded, spill file
|
|
110
|
+
# path) go out-of-band and never into the markdown body.
|
|
111
|
+
from .markdown import make_read_markdown
|
|
112
|
+
g["read_markdown"] = make_read_markdown(handle, g)
|
|
104
113
|
# Agent-editable layer last, so helpers can call core primitives.
|
|
105
114
|
_load_agent_helpers(g)
|
|
106
115
|
return g
|