browserwright 0.8.2__tar.gz → 0.9.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. {browserwright-0.8.2/src/browserwright.egg-info → browserwright-0.9.0}/PKG-INFO +2 -1
  2. {browserwright-0.8.2 → browserwright-0.9.0}/pyproject.toml +26 -1
  3. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/__init__.py +2 -1
  4. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/cli.py +152 -0
  5. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/extension_upstream.py +12 -0
  6. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/errors.py +28 -0
  7. browserwright-0.9.0/src/browserwright/repl/_md_convert.py +101 -0
  8. browserwright-0.9.0/src/browserwright/repl/_md_normalize.py +234 -0
  9. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/repl/_namespace.py +9 -0
  10. browserwright-0.9.0/src/browserwright/repl/markdown.py +260 -0
  11. browserwright-0.9.0/src/browserwright/repl/vendor/README.md +44 -0
  12. browserwright-0.9.0/src/browserwright/repl/vendor/readability.js +2812 -0
  13. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/session_runtime.py +12 -3
  14. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/skill_runtime.md +49 -8
  15. {browserwright-0.8.2 → browserwright-0.9.0/src/browserwright.egg-info}/PKG-INFO +2 -1
  16. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright.egg-info/SOURCES.txt +5 -0
  17. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright.egg-info/requires.txt +1 -0
  18. {browserwright-0.8.2 → browserwright-0.9.0}/LICENSE +0 -0
  19. {browserwright-0.8.2 → browserwright-0.9.0}/README.md +0 -0
  20. {browserwright-0.8.2 → browserwright-0.9.0}/setup.cfg +0 -0
  21. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/__main__.py +0 -0
  22. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/_executor/__init__.py +0 -0
  23. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/_executor/__main__.py +0 -0
  24. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/_executor/client.py +0 -0
  25. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/_executor/process.py +0 -0
  26. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/_executor/protocol.py +0 -0
  27. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/cdp.py +0 -0
  28. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/__init__.py +0 -0
  29. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/_ipc.py +0 -0
  30. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/_net.py +0 -0
  31. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/_rpc.py +0 -0
  32. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/_stale.py +0 -0
  33. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/backends/__init__.py +0 -0
  34. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/backends/base.py +0 -0
  35. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/backends/cdp.py +0 -0
  36. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/backends/extension.py +0 -0
  37. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/cli.py +0 -0
  38. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/config.py +0 -0
  39. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/doctor.py +0 -0
  40. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/errors.py +0 -0
  41. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/launch_chrome.py +0 -0
  42. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/launchagent.py +0 -0
  43. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/observability.py +0 -0
  44. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/platforms.py +0 -0
  45. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/probe.py +0 -0
  46. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/relay_status.py +0 -0
  47. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/resolver.py +0 -0
  48. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/__init__.py +0 -0
  49. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/daemon.py +0 -0
  50. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/executor_registry.py +0 -0
  51. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/facade.py +0 -0
  52. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/facade_extension.py +0 -0
  53. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/listener.py +0 -0
  54. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/proxy.py +0 -0
  55. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/relay.py +0 -0
  56. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/state.py +0 -0
  57. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/status.py +0 -0
  58. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/upstream.py +0 -0
  59. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/server/verbs.py +0 -0
  60. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/supervise.py +0 -0
  61. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/daemon/userscripts.py +0 -0
  62. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/discovery.py +0 -0
  63. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/health.py +0 -0
  64. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/install.py +0 -0
  65. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/memory/__init__.py +0 -0
  66. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/memory/_lock.py +0 -0
  67. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/memory/_md.py +0 -0
  68. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/memory/_yaml.py +0 -0
  69. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/memory/global_mem.py +0 -0
  70. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/memory/site_mem.py +0 -0
  71. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/mode_b_client.py +0 -0
  72. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/output_schema.py +0 -0
  73. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/primitives/__init__.py +0 -0
  74. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/primitives/discovery_api.py +0 -0
  75. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/primitives/http.py +0 -0
  76. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/primitives/site.py +0 -0
  77. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/repl/__init__.py +0 -0
  78. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/repl/_smart_goto.py +0 -0
  79. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/repl/inline.py +0 -0
  80. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/repl/playwright_handle.py +0 -0
  81. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/repl/snapshot.py +0 -0
  82. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/session.py +0 -0
  83. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/session_create.py +0 -0
  84. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/session_ctx.py +0 -0
  85. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/session_registry.py +0 -0
  86. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/github.com/SKILL.md +0 -0
  87. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/github.com/memory.md +0 -0
  88. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/github.com/tasks/list_issues.py +0 -0
  89. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/google.com/SKILL.md +0 -0
  90. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/google.com/memory.md +0 -0
  91. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/google.com/tasks/search.py +0 -0
  92. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/producthunt.com/SKILL.md +0 -0
  93. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/producthunt.com/memory.md +0 -0
  94. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/producthunt.com/tasks/today.py +0 -0
  95. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/wikipedia.org/SKILL.md +0 -0
  96. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/wikipedia.org/memory.md +0 -0
  97. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/wikipedia.org/tasks/lookup.py +0 -0
  98. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/ycombinator.com/SKILL.md +0 -0
  99. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/ycombinator.com/memory.md +0 -0
  100. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/site_skills_starter/ycombinator.com/tasks/front_page.py +0 -0
  101. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/skill_doc.py +0 -0
  102. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/task_runner.py +0 -0
  103. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright/version.py +0 -0
  104. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright.egg-info/dependency_links.txt +0 -0
  105. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright.egg-info/entry_points.txt +0 -0
  106. {browserwright-0.8.2 → browserwright-0.9.0}/src/browserwright.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: browserwright
3
- Version: 0.8.2
3
+ Version: 0.9.0
4
4
  Summary: Browserwright — let AI/code agents drive a real or isolated browser and author userscripts. Single package: the agent-facing REPL/site-skills/memory layer plus the bundled browser-resolving daemon (CDP proxy + extension/cloud backends).
5
5
  License-Expression: AGPL-3.0-only
6
6
  Requires-Python: >=3.11
@@ -10,6 +10,7 @@ Requires-Dist: pillow==12.2.0
10
10
  Requires-Dist: httpx>=0.27
11
11
  Requires-Dist: playwright>=1.60.0
12
12
  Requires-Dist: pyyaml>=6.0.3
13
+ Requires-Dist: html-to-markdown<4,>=3.10.6
13
14
  Provides-Extra: ux
14
15
  Requires-Dist: rich>=13; extra == "ux"
15
16
  Dynamic: license-file
@@ -15,7 +15,7 @@ name = "browserwright"
15
15
  # stamping regex is `^(version\s*=\s*)["'][^"']+["']\s*$` and CI aborts unless
16
16
  # it matches exactly once. `tests/skill/test_release_versioning.py` enforces all
17
17
  # of that before a tag is ever pushed. See RELEASING.md.
18
- version = "0.8.2"
18
+ version = "0.9.0"
19
19
  description = "Browserwright — let AI/code agents drive a real or isolated browser and author userscripts. Single package: the agent-facing REPL/site-skills/memory layer plus the bundled browser-resolving daemon (CDP proxy + extension/cloud backends)."
20
20
  requires-python = ">=3.11"
21
21
  license = "AGPL-3.0-only"
@@ -30,6 +30,26 @@ dependencies = [
30
30
  # daemon-resolved Chrome and does not need Playwright's bundled browser.
31
31
  "playwright>=1.60.0",
32
32
  "pyyaml>=6.0.3",
33
+ # HTML -> Markdown for the `read_markdown()` content view (ADR-0006/0007).
34
+ # Rust core + PyO3, and it pulls in NOTHING transitively — which is why it
35
+ # beat `markdownify` (bs4 + soupsieve) and `trafilatura` (lxml + a 32 MB
36
+ # babel locale tree) here. It is also the only converter measured that
37
+ # carries `class="language-*"` into a fenced ```lang block and keeps
38
+ # `rowspan` cells, both of which matter on the documentation pages this
39
+ # view exists to read.
40
+ #
41
+ # We deliberately track upstream rather than freeze: the project is running
42
+ # a systematic markdown-fidelity campaign (its own "Phase Z..HH" feat
43
+ # series) that keeps improving exactly what we depend on. The safety net for
44
+ # that is `tests/skill/test_markdown_golden.py` — golden files turn an
45
+ # upstream byte-level output change into a CI failure instead of a silent
46
+ # shift in what every agent reads. ADR-0007 makes those tests a precondition
47
+ # of this `>=`, not an optional extra.
48
+ #
49
+ # The `<4` cap is not conservatism about features: 3.0.0 removed the v1
50
+ # `convert_to_markdown()` entry point outright, so a major bump can break us
51
+ # at import time. Raise it deliberately, with the golden files as the diff.
52
+ "html-to-markdown>=3.10.6,<4",
33
53
  ]
34
54
 
35
55
  [project.optional-dependencies]
@@ -58,6 +78,11 @@ browserwright = [
58
78
  "site_skills_starter/**/*.md",
59
79
  "site_skills_starter/**/*.py",
60
80
  "skill_runtime.md",
81
+ # Vendored Mozilla Readability, injected as SOURCE TEXT via page.evaluate
82
+ # (add_script_tag is blocked by page CSP; page.evaluate is not). It has to
83
+ # ship as a real file for that reason — see repl/vendor/README.md.
84
+ "repl/vendor/*.js",
85
+ "repl/vendor/*.md",
61
86
  ]
62
87
 
63
88
  [tool.setuptools.exclude-package-data]
@@ -42,6 +42,7 @@ from .errors import ( # noqa: F401
42
42
  NeedsUserConfirm,
43
43
  NetworkError,
44
44
  PageLoadFailed,
45
+ UnsupportedContentType,
45
46
  )
46
47
  from .primitives.discovery_api import ( # noqa: F401
47
48
  list_site_skills,
@@ -69,7 +70,7 @@ EXPORTS = [
69
70
  # errors
70
71
  "BrowserwrightError", "PageLoadFailed", "ElementNotFound", "AuthWall",
71
72
  "Captcha", "NetworkError", "DaemonUnavailable", "CDPError",
72
- "NeedsUserConfirm",
73
+ "NeedsUserConfirm", "UnsupportedContentType",
73
74
  ]
74
75
 
75
76
  __all__ = EXPORTS
@@ -38,6 +38,11 @@ Usage:
38
38
  browserwright whoami --session=ID
39
39
  browserwright userscript {push|list|remove|toggle|logs} ...
40
40
 
41
+ browserwright markdown <url> [--mode=auto|article|full] [--backend=extension|cdp]
42
+ [--out=PATH] [--max-chars=N]
43
+ One page as Markdown. Creates and tears down its own session, so it takes
44
+ no -s. Absolute links, shadow DOM flattened in, HTML only.
45
+
41
46
  browserwright -s <session-id> task <site>/<name> [--key=value ...] [--isolated]
42
47
  browserwright list-tasks [--site SITE] [--query Q] [--json]
43
48
 
@@ -418,6 +423,149 @@ def _cmd_task(args: list[str], *, session_id: Optional[str] = None) -> int:
418
423
  return 0
419
424
 
420
425
 
426
+ MARKDOWN_HELP = """Usage:
427
+ browserwright markdown <url> [--mode=auto|article|full] [--backend=extension|cdp]
428
+ [--out=PATH] [--max-chars=N] [--name=LABEL]
429
+
430
+ Fetch one page and print it as Markdown. Owns its whole lifecycle: it creates a
431
+ throwaway session, navigates, converts, and tears the session down again — so
432
+ unlike every other browser-driving command it takes no -s/--session.
433
+
434
+ Links come out absolute, open shadow roots are flattened in, and same-origin
435
+ iframes are inlined. Only HTML is converted; anything else is refused with its
436
+ Content-Type rather than silently returning an empty-looking result.
437
+
438
+ Flags:
439
+ --mode=auto (default) extract the main content; if extraction collapses,
440
+ fall back to the page with nav/aside/footer/forms removed
441
+ --mode=article force extraction
442
+ --mode=full the page verbatim — use it when you want the navigation,
443
+ a form, or every link
444
+ --backend extension (default, the user's real Chrome) | cdp
445
+ --out=PATH write the full Markdown here instead of a temp file
446
+ --max-chars=N cap what is printed (default 8000; 0 prints everything).
447
+ The FULL text is always written to the file either way.
448
+ --name=LABEL session label; shows as the Chrome tab group title
449
+ """
450
+
451
+ # Runs inside the session's executor, which is the only place a live `page`
452
+ # exists. Values arrive through the request environment (names only cross the
453
+ # CLI boundary, never values — same discipline as `--env`), and the markdown
454
+ # rides back on disk rather than through the executor's 10000-char console cap.
455
+ _MARKDOWN_CODE = """
456
+ import os
457
+ from browserwright.repl.markdown import render_page_markdown
458
+
459
+ page.goto(os.environ["BW_MD_URL"])
460
+ _r = render_page_markdown(page, mode=os.environ["BW_MD_MODE"])
461
+ with open(os.environ["BW_MD_OUT"], "w", encoding="utf-8") as _f:
462
+ _f.write(_r.markdown)
463
+ _bw_warn("markdown: rendered as %s (%d chars)" % (_r.mode_used, len(_r.markdown)))
464
+ for _n in _r.notes:
465
+ _bw_warn("markdown: " + _n)
466
+ """
467
+
468
+
469
+ def _cmd_markdown(args: list[str]) -> int:
470
+ """``browserwright markdown <url>`` — one page, as Markdown (ADR-0006).
471
+
472
+ The one browser-driving command with no session argument: it mints a
473
+ throwaway one, uses it, and ends it. Teardown runs in a `finally` because a
474
+ leaked session leaves a tab group behind in the user's real Chrome (see
475
+ issue #53 for why that path is worth watching).
476
+ """
477
+ if not args or args[0] in {"-h", "--help"}:
478
+ sys.stdout.write(MARKDOWN_HELP)
479
+ return 0 if args else 1
480
+
481
+ url = args[0]
482
+ if url.startswith("-"):
483
+ print(f"usage error: expected a URL, got {url!r}", file=sys.stderr)
484
+ print(MARKDOWN_HELP, file=sys.stderr)
485
+ return 1
486
+ kw = _parse_kv_args(args[1:])
487
+
488
+ mode = str(kw.get("mode", "auto"))
489
+ from .repl.markdown import DEFAULT_MAX_CHARS, MODES, spill_path
490
+
491
+ if mode not in MODES:
492
+ print(f"usage error: --mode must be one of {'|'.join(MODES)}, "
493
+ f"got {mode!r}", file=sys.stderr)
494
+ return 1
495
+ backend = str(kw.get("backend", "extension"))
496
+ if backend not in ("extension", "cdp"):
497
+ print(f"usage error: --backend must be extension or cdp, got "
498
+ f"{backend!r}", file=sys.stderr)
499
+ return 1
500
+ try:
501
+ max_chars = int(kw.get("max-chars", DEFAULT_MAX_CHARS))
502
+ except (TypeError, ValueError):
503
+ print("usage error: --max-chars must be an integer (0 = no cap)",
504
+ file=sys.stderr)
505
+ return 1
506
+
507
+ out = kw.get("out")
508
+ keep_file = out is not None
509
+ out_path = str(out) if keep_file else spill_path(url)
510
+
511
+ import contextlib
512
+
513
+ from . import session_create
514
+ from .errors import BrowserwrightError
515
+ from .repl import inline
516
+ from .session_ctx import resolve_session_or_env
517
+
518
+ try:
519
+ sid = session_create.new(
520
+ backend=backend,
521
+ # `cdp` has no meaning without an owner: it must either launch a
522
+ # browser or attach to one, and a throwaway session has nobody to
523
+ # attach to. NOTE this launches a real Chrome, which on macOS takes
524
+ # the active window — the default backend is `extension` precisely
525
+ # so the common path never does that.
526
+ create=(backend == "cdp"),
527
+ attach=None,
528
+ name=str(kw.get("name", "markdown")),
529
+ )
530
+ except (ValueError, BrowserwrightError) as e:
531
+ print(str(e), file=sys.stderr)
532
+ return 1
533
+
534
+ try:
535
+ rc = inline.run_code(
536
+ _MARKDOWN_CODE,
537
+ session_id=sid,
538
+ env={"BW_MD_URL": url, "BW_MD_OUT": out_path, "BW_MD_MODE": mode},
539
+ )
540
+ if rc != 0:
541
+ return rc
542
+ try:
543
+ text = Path(out_path).read_text(encoding="utf-8")
544
+ except OSError as e:
545
+ print(f"markdown: could not read rendered output: {e}",
546
+ file=sys.stderr)
547
+ return 3
548
+ if max_chars > 0 and len(text) > max_chars:
549
+ from .repl.snapshot import _truncate_lines
550
+
551
+ print(f"[markdown] truncated to {max_chars} of {len(text)} chars; "
552
+ f"full text: {out_path}", file=sys.stderr)
553
+ sys.stdout.write(_truncate_lines(text, max_chars) + "\n")
554
+ keep_file = True # the caller needs it to see the rest
555
+ else:
556
+ sys.stdout.write(text if text.endswith("\n") else text + "\n")
557
+ return 0
558
+ finally:
559
+ if not keep_file:
560
+ with contextlib.suppress(OSError):
561
+ Path(out_path).unlink()
562
+ try:
563
+ session_create.end(resolve_session_or_env(sid))
564
+ except Exception as e: # noqa: BLE001 — never mask the real result
565
+ print(f"[markdown] warning: could not end throwaway session {sid}: "
566
+ f"{e}", file=sys.stderr)
567
+
568
+
421
569
  def _cmd_doctor(args: list[str]) -> int:
422
570
  """A4: a ``{status, message, fix}`` check table.
423
571
 
@@ -890,6 +1038,10 @@ def main(argv: Optional[list[str]] = None) -> None:
890
1038
  sys.exit(_cmd_index(rest))
891
1039
  if cmd == "memory":
892
1040
  sys.exit(_cmd_memory(rest))
1041
+ # Deliberately NOT under the -s branch: this is the one browser-driving
1042
+ # command that owns its own throwaway session (ADR-0006).
1043
+ if cmd == "markdown":
1044
+ sys.exit(_cmd_markdown(rest))
893
1045
  if cmd == "session":
894
1046
  sys.exit(_cmd_session(rest, session_id=global_session))
895
1047
  if cmd == "whoami":
@@ -1242,6 +1242,18 @@ class ExtensionUpstream:
1242
1242
  group_id: int) -> dict:
1243
1243
  resolved_group_id, info = await self._resolve_session_group(
1244
1244
  session_id, group_id)
1245
+ # The adapter-memory fast path returns ``(gid, None)`` — the group
1246
+ # query is deferred to the caller. Mirror `_group_member_tabs` and
1247
+ # re-query by the resolved id before concluding the group is empty
1248
+ # (e2e recovery test: an in-daemon session that just opened a tab
1249
+ # hits the memory path and would otherwise be reported "group
1250
+ # missing" despite a live group with tabs).
1251
+ if info is None and isinstance(resolved_group_id, int) and resolved_group_id >= 0:
1252
+ query_generation = self._relay_generation()
1253
+ info = await self._relay.query_group_tabs(group_id=resolved_group_id)
1254
+ if query_generation != self._relay_generation():
1255
+ raise RuntimeError(
1256
+ "extension reconnected while resolving group membership")
1245
1257
  if not info or not info.get("tabs"):
1246
1258
  raise RuntimeError(
1247
1259
  f"no recoverable tabs for group id {group_id} "
@@ -75,6 +75,34 @@ class ElementNotFound(BrowserwrightError):
75
75
  super().__init__(f"element not found: {selector!r} after {timeout}s", fix=fix)
76
76
 
77
77
 
78
+ class UnsupportedContentType(BrowserwrightError):
79
+ """The page is not HTML, so there is nothing to convert to Markdown.
80
+
81
+ Raised loudly on purpose (ADR-0007). A PDF is the motivating case: Chrome's
82
+ built-in viewer renders a REAL DOM (an ``<embed>`` shell), so a best-effort
83
+ conversion would succeed and return a short, plausible-looking result that
84
+ contains none of the document — a silent empty answer, which is the one
85
+ failure mode the markdown path is built to make impossible.
86
+
87
+ browserwright deliberately does not grow a document pipeline. Route the URL
88
+ to whatever handles that type and keep browserwright for HTML.
89
+ """
90
+
91
+ exit_code = 3
92
+ default_fix = (
93
+ "this endpoint only converts HTML; route non-HTML types "
94
+ "(PDF, images, binaries) to a handler for that type"
95
+ )
96
+
97
+ def __init__(self, url: str = "", content_type: str = "", fix: str = ""):
98
+ self.url, self.content_type = url, content_type
99
+ super().__init__(
100
+ f"not HTML, refusing to convert: {url} (Content-Type: "
101
+ f"{content_type or 'unknown'})",
102
+ fix=fix,
103
+ )
104
+
105
+
78
106
  class AuthWall(BrowserwrightError):
79
107
  exit_code = 4
80
108
  default_fix = "stop and ask the user to log in; do not type credentials from a screenshot"
@@ -0,0 +1,101 @@
1
+ """The one and only place HTML becomes Markdown.
2
+
3
+ ADR-0007 requires this seam: we track `html-to-markdown` with a `>=` rather
4
+ than freezing it, so the day we swap converters — or the day upstream changes a
5
+ default out from under us — exactly one function has to change, and
6
+ ``tests/skill/test_markdown_golden.py`` is what makes the change visible.
7
+
8
+ Nothing else in the tree may import ``html_to_markdown`` directly.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ from typing import Any
13
+
14
+ # Built once per variant. `ConversionOptions` is a plain frozen options object,
15
+ # so module singletons are safe and save rebuilding 44 fields per page.
16
+ _OPTIONS: dict[bool, Any] = {}
17
+
18
+
19
+ def _options(strip_chrome: bool) -> Any:
20
+ if strip_chrome not in _OPTIONS:
21
+ from html_to_markdown import ConversionOptions, PreprocessingOptions
22
+
23
+ # Two tiers of "remove things", and the difference is the whole point:
24
+ #
25
+ # strip_chrome=False -> remove NOTHING. The verbatim page.
26
+ # strip_chrome=True -> remove KNOWN chrome by tag/class (`<nav>`,
27
+ # `<aside>`, `<footer>`, forms, `.sidebar`…).
28
+ #
29
+ # Neither one scores or guesses which block is the article — that is
30
+ # Readability's job, upstream of here, and the reason this second tier
31
+ # exists at all is that Readability's guessing can wipe a page out
32
+ # (a GitHub issue page: 118 links -> 0, 2866 tokens -> 92). Tag-based
33
+ # removal cannot do that: it deletes the elements it recognizes and
34
+ # keeps everything else, so it is the honest fallback when extraction
35
+ # collapses — the reader wanted the body text, not the site furniture.
36
+ #
37
+ # Measured on 3.10.6 with nav/header/main/aside/footer/.sidebar/form:
38
+ # enabled=False keeps all 8 markers
39
+ # preset="standard" drops only <nav> and <form>
40
+ # preset="aggressive" keeps header + body text + body links; drops the
41
+ # rest of the furniture
42
+ pre = (
43
+ PreprocessingOptions(enabled=True, preset="aggressive")
44
+ if strip_chrome
45
+ else PreprocessingOptions(enabled=False)
46
+ )
47
+ _OPTIONS[strip_chrome] = ConversionOptions(
48
+ # Never left at its default. `preprocessing=None` is NOT "off" — it
49
+ # means "use the built-in default", which is
50
+ # `enabled=True, remove_navigation=True, remove_forms=True`:
51
+ # '<nav><a href="/n">nav link</a></nav><p>body</p><form>…</form>'
52
+ # default -> 'body\n'
53
+ # enabled=False -> '[nav link](/n)\n\nbody\n\nQ\n'
54
+ # So out of the box this library quietly deletes navigation and
55
+ # forms on EVERY conversion, including the one the caller asked to
56
+ # be verbatim. Removal has to be our decision, driven by `mode`, and
57
+ # announced — never a converter default nobody chose.
58
+ preprocessing=pre,
59
+ # Default is False, and leaving it there pads every separator row
60
+ # out to the widest cell in the column: a wide Wikipedia table
61
+ # measured 984,849 chars vs 359,598 with this on. That single flag
62
+ # is the whole of this library's "token-hungry" reputation.
63
+ compact_tables=True,
64
+ # Default is True, which prepends a YAML front-matter block built
65
+ # from <title>/<meta> whenever the document has a <head>. The agent
66
+ # asked for page content, not for a document envelope, and every
67
+ # downstream consumer would have to strip it (ADR-0007: metadata
68
+ # goes out-of-band, never into the body).
69
+ extract_metadata=False,
70
+ # Already the upstream default today. Pinned explicitly anyway
71
+ # because we track upstream with `>=`: ATX (`# h1`) vs setext
72
+ # (`h1\n===`) changes the bytes of every heading, and a default
73
+ # flip should break a golden file, not silently reshape output.
74
+ heading_style="atx",
75
+ )
76
+ return _OPTIONS[strip_chrome]
77
+
78
+
79
+ def convert_html(html: str, *, strip_chrome: bool = False) -> str:
80
+ """Convert an HTML string to Markdown. Returns "" for empty input.
81
+
82
+ ``strip_chrome=True`` additionally removes recognized page furniture
83
+ (nav / aside / footer / forms / sidebar-ish class names) by tag and class,
84
+ never by scoring. Use it for the fallback path where the reader wants body
85
+ text; leave it off wherever the caller asked for the page verbatim, and off
86
+ for a Readability fragment, which has already been stripped.
87
+
88
+ The caller owns the empty-output decision. This function does NOT guard
89
+ against the upstream defect where a stray ``<td>``/``<th>`` outside a table
90
+ silently collapses its whole subtree to ``""`` (verified against 3.10.6:
91
+ ``<div><td>hello</td></div>`` -> ``""``). That defect cannot fire on
92
+ browser-serialized HTML, which is always well-formed — only on hand-built
93
+ fragments, i.e. the Readability path — so the guard lives where the
94
+ fragment is produced, as the "empty -> fall back to the full page" rule of
95
+ ADR-0007.
96
+ """
97
+ if not html or not html.strip():
98
+ return ""
99
+ from html_to_markdown import convert
100
+
101
+ return convert(html, _options(strip_chrome)).content
@@ -0,0 +1,234 @@
1
+ """The in-page half of the markdown pipeline: normalize the live DOM to HTML.
2
+
3
+ This exists because ``page.content()`` silently loses two things Markdown needs,
4
+ and neither can be recovered on the Python side (ADR-0007):
5
+
6
+ - **open shadow roots are not serialized at all.** Verified in this repo: a page
7
+ with a declarative ``<template shadowrootmode="open">`` reports ``shadowRoot``
8
+ present, in-page JS reads its text fine, and the string from
9
+ ``page.content()`` does not contain it.
10
+ - **relative URLs stay relative.** No Python converter resolves ``<base>``. The
11
+ browser resolves per document — including separately inside each iframe.
12
+
13
+ So this script runs where the DOM is, rebuilds a detached copy with those two
14
+ problems fixed, and hands back plain HTML strings.
15
+
16
+ Why ``page.evaluate`` and not ``add_script_tag``
17
+ ------------------------------------------------
18
+ Verified in this repo against ``default-src 'none'; script-src 'none'`` with
19
+ ``bypass_csp=False``: the page's own inline ``<script>`` does NOT run (CSP is
20
+ being enforced), while ``page.evaluate("new Function('return 41+1')()")``
21
+ returns ``42`` and ``add_script_tag`` raises. CDP's ``Runtime.evaluate`` carries
22
+ ``allowUnsafeEvalBlockedByCSP``, so protocol-driven evaluation is exempt where
23
+ page-driven script injection is not.
24
+
25
+ Three consequences the script below obeys:
26
+
27
+ - **Stay synchronous.** The CSP exemption is a toggle held for the duration of
28
+ the protocol call; a continuation scheduled from inside it (``setTimeout``,
29
+ ``await``) loses it. Everything here is one synchronous pass.
30
+ - **Never assign ``innerHTML``, never use ``DOMParser``.** Those are gated by
31
+ Trusted Types, which ``page.evaluate`` does NOT bypass — it runs in the main
32
+ world. *Reading* ``innerHTML``/``outerHTML`` and calling ``cloneNode`` /
33
+ ``importNode`` are unaffected, which is why the copy is built with node APIs
34
+ and HTML is only ever read out.
35
+ - **Never mutate the live page.** Everything is built detached, and Readability
36
+ (which rewrites whatever document it is handed) only ever sees a scratch
37
+ document.
38
+
39
+ And do not "fix" any of this by turning on ``bypass_csp``: that lets the page's
40
+ previously-blocked inline scripts run and stops Trusted Types being enforced,
41
+ i.e. it changes the very DOM we are here to capture.
42
+ """
43
+ from __future__ import annotations
44
+
45
+ from functools import lru_cache
46
+ from pathlib import Path
47
+
48
+ _VENDOR = Path(__file__).resolve().parent / "vendor"
49
+
50
+
51
+ @lru_cache(maxsize=1)
52
+ def _readability_source() -> str:
53
+ """Mozilla Readability 0.6.0, vendored. See ``vendor/README.md``."""
54
+ return (_VENDOR / "readability.js").read_text(encoding="utf-8")
55
+
56
+
57
+ # The body of the injected function. Assembled with the Readability source in
58
+ # `build_script()` below, because the extraction pass needs it in scope.
59
+ _NORMALIZE_BODY = r"""
60
+ // Elements dropped wholesale: no readable content, pure token cost.
61
+ // `svg`/`canvas` are decorative here - when they carry an accessible name it
62
+ // lives on an ancestor, which survives. `template` is author-declared inert
63
+ // content; declarative shadow DOM is consumed by the parser long before this
64
+ // runs and surfaces as a real shadowRoot, handled below.
65
+ const DROP = new Set([
66
+ "SCRIPT", "STYLE", "NOSCRIPT", "TEMPLATE", "SVG", "CANVAS",
67
+ "LINK", "META", "BASE",
68
+ ]);
69
+ // Attributes whose value is a URL we want absolute in the output.
70
+ const URL_ATTRS = { A: "href", AREA: "href", IMG: "src", SOURCE: "src" };
71
+ const MAX_DEPTH = 128;
72
+
73
+ const stats = {
74
+ shadowRoots: 0,
75
+ sameOriginFrames: 0,
76
+ crossOriginFrames: 0,
77
+ linksTotal: 0,
78
+ };
79
+
80
+ const isHidden = (el) => {
81
+ if (el.hasAttribute("hidden")) return true;
82
+ if (el.getAttribute("aria-hidden") === "true") return true;
83
+ return false;
84
+ };
85
+
86
+ // Resolve against the *owning* document's base URI, which differs per frame.
87
+ // Reading the IDL property (`a.href`) would resolve nodes imported from an
88
+ // iframe against the top document instead, so resolve explicitly.
89
+ const absolutize = (src, clone, baseURI) => {
90
+ const attr = URL_ATTRS[src.nodeName.toUpperCase()];
91
+ if (!attr) return;
92
+ const raw = src.getAttribute(attr);
93
+ if (!raw) return;
94
+ if (attr === "href") stats.linksTotal += 1;
95
+ try {
96
+ clone.setAttribute(attr, new URL(raw, baseURI).href);
97
+ } catch (e) {
98
+ /* mailto:, javascript:, malformed - keep the author's value */
99
+ }
100
+ };
101
+
102
+ // Build a detached copy of `node`. Returns null when the node is dropped.
103
+ const build = (node, baseURI, depth) => {
104
+ if (depth > MAX_DEPTH) return null;
105
+
106
+ if (node.nodeType === Node.TEXT_NODE) {
107
+ return node.data.trim() ? node.cloneNode(false) : null;
108
+ }
109
+ if (node.nodeType !== Node.ELEMENT_NODE) return null;
110
+
111
+ const name = node.nodeName.toUpperCase();
112
+ if (DROP.has(name)) return null;
113
+ if (isHidden(node)) return null;
114
+
115
+ if (name === "IFRAME") {
116
+ // A null contentDocument IS the same-origin test. Sandboxed frames can
117
+ // throw rather than return null, hence the guard.
118
+ let doc = null;
119
+ try { doc = node.contentDocument; } catch (e) { doc = null; }
120
+ if (!doc || !doc.body) {
121
+ stats.crossOriginFrames += 1;
122
+ return null;
123
+ }
124
+ stats.sameOriginFrames += 1;
125
+ // Inline the frame body at the iframe's position, wrapped so the
126
+ // converter sees a block boundary instead of splicing the frame's
127
+ // paragraphs into the parent's.
128
+ const holder = document.createElement("div");
129
+ for (const child of Array.from(doc.body.childNodes)) {
130
+ const built = build(child, doc.baseURI, depth + 1);
131
+ if (built) holder.appendChild(built);
132
+ }
133
+ return holder.childNodes.length ? holder : null;
134
+ }
135
+
136
+ // Shallow clone keeps attributes; children are rebuilt so we control
137
+ // shadow/slot/iframe expansion.
138
+ const clone = node.cloneNode(false);
139
+ absolutize(node, clone, baseURI);
140
+
141
+ // A SLOT renders its assigned nodes. Expanding it here is what makes a
142
+ // flattened shadow tree come out in composed-tree order rather than
143
+ // shadow-then-light.
144
+ if (name === "SLOT" && typeof node.assignedNodes === "function") {
145
+ const holder = document.createElement("div");
146
+ for (const child of node.assignedNodes({ flatten: true })) {
147
+ const built = build(child, baseURI, depth + 1);
148
+ if (built) holder.appendChild(built);
149
+ }
150
+ return holder.childNodes.length ? holder : null;
151
+ }
152
+
153
+ // Open shadow root: its content is what the user actually sees, and the
154
+ // light children are reached through the slots inside it. A CLOSED root is
155
+ // indistinguishable from no root here - `shadowRoot` is null either way -
156
+ // so closed roots are invisible to every approach, ours included, and we
157
+ // cannot even report their number. See ADR-0007.
158
+ const root = node.shadowRoot;
159
+ if (root) {
160
+ stats.shadowRoots += 1;
161
+ for (const child of Array.from(root.childNodes)) {
162
+ const built = build(child, baseURI, depth + 1);
163
+ if (built) clone.appendChild(built);
164
+ }
165
+ return clone;
166
+ }
167
+
168
+ for (const child of Array.from(node.childNodes)) {
169
+ const built = build(child, baseURI, depth + 1);
170
+ if (built) clone.appendChild(built);
171
+ }
172
+ return clone;
173
+ };
174
+
175
+ // A scratch document, so Readability has something it is allowed to destroy
176
+ // and so the normalized tree is a real document rather than a loose node.
177
+ const scratch = document.implementation.createHTMLDocument(document.title || "");
178
+ const normalized = document.body ? build(document.body, document.baseURI, 0) : null;
179
+ if (normalized) {
180
+ for (const child of Array.from(normalized.childNodes)) {
181
+ scratch.body.appendChild(scratch.importNode(child, true));
182
+ }
183
+ }
184
+
185
+ const result = {
186
+ fullHtml: scratch.body.innerHTML,
187
+ articleHtml: null,
188
+ articleLinks: 0,
189
+ title: document.title || "",
190
+ url: location.href,
191
+ contentType: document.contentType || "",
192
+ stats: stats,
193
+ };
194
+
195
+ if (opts && opts.extract) {
196
+ // Readability rewrites the document it is given, so hand it a copy of the
197
+ // scratch document - never `scratch` itself, which we still need above.
198
+ try {
199
+ const forReader = scratch.cloneNode(true);
200
+ const article = new Readability(forReader).parse();
201
+ if (article && article.content) {
202
+ result.articleHtml = article.content;
203
+ // Counted here rather than in Python: the caller's collapse test is
204
+ // about the DOM that came out, and re-parsing markdown to count links
205
+ // would measure the converter instead.
206
+ const probe = document.implementation.createHTMLDocument("");
207
+ probe.body.appendChild(probe.importNode(forReader.body, true));
208
+ result.articleLinks = probe.querySelectorAll("a[href]").length;
209
+ }
210
+ } catch (e) {
211
+ result.articleHtml = null; // treated as a collapse by the caller
212
+ }
213
+ }
214
+
215
+ return result;
216
+ """
217
+
218
+
219
+ def build_script(*, extract: bool) -> str:
220
+ """Assemble the full ``page.evaluate`` payload.
221
+
222
+ Returns the source of a single arrow function taking one options object.
223
+ The Readability source is inlined only when extraction is requested — it is
224
+ ~91 KB, and shipping it across the CDP boundary on every full-page read
225
+ would be pure overhead.
226
+ """
227
+ prelude = ""
228
+ if extract:
229
+ # Readability ends with `if (typeof module === "object") module.exports
230
+ # = Readability;`. In a page that leaked a bundler's `module` global
231
+ # (webpack/browserify do), that assignment would mutate the page. A
232
+ # local binding shadows it so the branch is never taken.
233
+ prelude = " var module = void 0;\n" + _readability_source() + "\n"
234
+ return "(opts) => {\n" + prelude + _NORMALIZE_BODY + "\n}"
@@ -101,6 +101,15 @@ def build_globals() -> dict[str, Any]:
101
101
  # to override.
102
102
  from .snapshot import make_snapshot
103
103
  g["snapshot"] = make_snapshot(handle)
104
+ # The second view (ADR-0006): `snapshot()` is the action view, this is the
105
+ # content view. Same rules — injected per heredoc, read-only, never in
106
+ # EXPORTS (a view needs the live `page`, which a module-level function
107
+ # cannot have). `g` is passed so the closure can find the executor's
108
+ # `_bw_warn` channel, which is injected into this same dict *after* we
109
+ # return; its notes (extraction path taken, iframes excluded, spill file
110
+ # path) go out-of-band and never into the markdown body.
111
+ from .markdown import make_read_markdown
112
+ g["read_markdown"] = make_read_markdown(handle, g)
104
113
  # Agent-editable layer last, so helpers can call core primitives.
105
114
  _load_agent_helpers(g)
106
115
  return g