lazycodex-ai 5.0.0-beta.82 → 5.0.0-beta.84

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (110) hide show
  1. package/README.ja.md +2 -2
  2. package/README.ko.md +2 -2
  3. package/README.md +4 -4
  4. package/README.ru.md +2 -2
  5. package/README.zh-cn.md +2 -2
  6. package/dist/cli/index.js +707 -146
  7. package/dist/cli/install-native/index.d.ts +4 -0
  8. package/dist/cli/install-native/plan.d.ts +11 -0
  9. package/dist/cli/install-native/run-native-install.d.ts +23 -0
  10. package/dist/cli/install-native-dev/index.d.ts +4 -0
  11. package/dist/cli/native-dev-platform-flag.d.ts +5 -0
  12. package/dist/cli/native-edition-hint.d.ts +10 -0
  13. package/dist/cli/types.d.ts +3 -2
  14. package/dist/cli-node/index.js +707 -146
  15. package/package.json +1 -1
  16. package/packages/omo-codex/plugin/.codex-plugin/plugin.json +1 -1
  17. package/packages/omo-codex/plugin/components/bootstrap/dist/cli.js +84 -11
  18. package/packages/omo-codex/plugin/components/bootstrap/hooks/hooks.json +1 -1
  19. package/packages/omo-codex/plugin/components/bootstrap/package.json +1 -1
  20. package/packages/omo-codex/plugin/components/comment-checker/hooks/hooks.json +1 -1
  21. package/packages/omo-codex/plugin/components/comment-checker/package.json +1 -1
  22. package/packages/omo-codex/plugin/components/git-bash/hooks/hooks.json +2 -2
  23. package/packages/omo-codex/plugin/components/git-bash/package.json +1 -1
  24. package/packages/omo-codex/plugin/components/lazycodex-executor-verify/hooks/hooks.json +1 -1
  25. package/packages/omo-codex/plugin/components/lazycodex-executor-verify/package.json +1 -1
  26. package/packages/omo-codex/plugin/components/lcx/skills/lcx-contribute-bug-fix/SKILL.md +1 -1
  27. package/packages/omo-codex/plugin/components/lcx/skills/lcx-doctor/SKILL.md +3 -3
  28. package/packages/omo-codex/plugin/components/lcx/skills/lcx-report-bug/SKILL.md +5 -5
  29. package/packages/omo-codex/plugin/components/lsp/dist/.omo-runtime-manifest.json +2 -2
  30. package/packages/omo-codex/plugin/components/lsp/hooks/hooks.json +2 -2
  31. package/packages/omo-codex/plugin/components/lsp/package.json +1 -1
  32. package/packages/omo-codex/plugin/components/rules/hooks/hooks.json +4 -4
  33. package/packages/omo-codex/plugin/components/rules/package.json +1 -1
  34. package/packages/omo-codex/plugin/components/teammode/hooks/hooks.json +1 -1
  35. package/packages/omo-codex/plugin/components/teammode/package.json +1 -1
  36. package/packages/omo-codex/plugin/components/telemetry/hooks/hooks.json +1 -1
  37. package/packages/omo-codex/plugin/components/telemetry/package.json +1 -1
  38. package/packages/omo-codex/plugin/components/ultrawork/hooks/hooks.json +1 -1
  39. package/packages/omo-codex/plugin/components/ultrawork/package.json +1 -1
  40. package/packages/omo-codex/plugin/components/ulw-execute-continuation/hooks/hooks.json +1 -1
  41. package/packages/omo-codex/plugin/components/ulw-execute-continuation/package.json +1 -1
  42. package/packages/omo-codex/plugin/components/ulw-loop/dist/cli.js +24 -10
  43. package/packages/omo-codex/plugin/components/ulw-loop/dist/spawn-guard.js +8 -6
  44. package/packages/omo-codex/plugin/components/ulw-loop/dist/surface.d.ts +2 -0
  45. package/packages/omo-codex/plugin/components/ulw-loop/dist/surface.js +21 -4
  46. package/packages/omo-codex/plugin/components/ulw-loop/hooks/hooks.json +5 -5
  47. package/packages/omo-codex/plugin/components/ulw-loop/package.json +1 -1
  48. package/packages/omo-codex/plugin/components/ulw-loop/src/spawn-guard.ts +10 -5
  49. package/packages/omo-codex/plugin/components/ulw-loop/src/surface.ts +24 -6
  50. package/packages/omo-codex/plugin/components/ulw-loop/test/spawn-guard-native-reviewer.test.ts +207 -0
  51. package/packages/omo-codex/plugin/components/ulw-loop/test/spawn-guard.test.ts +6 -4
  52. package/packages/omo-codex/plugin/components/ulw-loop/test/surface.test.ts +43 -4
  53. package/packages/omo-codex/plugin/hooks/post-compact-resetting-git-bash-mcp-reminder.json +1 -1
  54. package/packages/omo-codex/plugin/hooks/post-compact-resetting-lsp-diagnostics-cache.json +1 -1
  55. package/packages/omo-codex/plugin/hooks/post-compact-resetting-project-rule-cache.json +1 -1
  56. package/packages/omo-codex/plugin/hooks/post-tool-use-checking-comments.json +1 -1
  57. package/packages/omo-codex/plugin/hooks/post-tool-use-checking-lsp-diagnostics.json +1 -1
  58. package/packages/omo-codex/plugin/hooks/post-tool-use-checking-thread-title-hygiene.json +1 -1
  59. package/packages/omo-codex/plugin/hooks/post-tool-use-matching-project-rules.json +1 -1
  60. package/packages/omo-codex/plugin/hooks/post-tool-use-recording-spawn-admission.json +1 -1
  61. package/packages/omo-codex/plugin/hooks/pre-tool-use-enforcing-unlimited-goal-budget.json +1 -1
  62. package/packages/omo-codex/plugin/hooks/pre-tool-use-guarding-ulw-loop-spawns.json +1 -1
  63. package/packages/omo-codex/plugin/hooks/pre-tool-use-recommending-git-bash-mcp.json +1 -1
  64. package/packages/omo-codex/plugin/hooks/session-start-checking-auto-update.json +1 -1
  65. package/packages/omo-codex/plugin/hooks/session-start-checking-bootstrap-provisioning.json +1 -1
  66. package/packages/omo-codex/plugin/hooks/session-start-loading-project-rules.json +1 -1
  67. package/packages/omo-codex/plugin/hooks/session-start-recording-session-telemetry.json +1 -1
  68. package/packages/omo-codex/plugin/hooks/stop-checking-ulw-execute-continuation.json +1 -1
  69. package/packages/omo-codex/plugin/hooks/stop-checking-ulw-loop-resume.json +1 -1
  70. package/packages/omo-codex/plugin/hooks/subagent-stop-verifying-lazycodex-executor-evidence.json +1 -1
  71. package/packages/omo-codex/plugin/hooks/user-prompt-submit-checking-ultrawork-trigger.json +1 -1
  72. package/packages/omo-codex/plugin/hooks/user-prompt-submit-checking-ulw-loop-steering.json +1 -1
  73. package/packages/omo-codex/plugin/hooks/user-prompt-submit-loading-project-rules.json +1 -1
  74. package/packages/omo-codex/plugin/package-lock.json +12 -12
  75. package/packages/omo-codex/plugin/package.json +1 -1
  76. package/packages/omo-codex/plugin/skills/browser/ATTRIBUTION.md +25 -0
  77. package/packages/omo-codex/plugin/skills/browser/SKILL.md +89 -0
  78. package/packages/omo-codex/plugin/skills/browser/agents/openai.yaml +2 -0
  79. package/packages/omo-codex/plugin/skills/browser/references/commands.md +90 -0
  80. package/packages/omo-codex/plugin/skills/browser/references/install.md +71 -0
  81. package/packages/omo-codex/plugin/skills/browser/references/owned-engine/README.md +42 -0
  82. package/packages/omo-codex/plugin/skills/browser/references/owned-engine/frames-and-humans.md +39 -0
  83. package/packages/omo-codex/plugin/skills/browser/references/owned-engine/ladder.md +59 -0
  84. package/packages/omo-codex/plugin/skills/browser/references/owned-engine/network.md +42 -0
  85. package/packages/omo-codex/plugin/skills/browser/references/recipes/1password.md +59 -0
  86. package/packages/omo-codex/plugin/skills/browser/references/remote.md +24 -0
  87. package/packages/omo-codex/plugin/skills/browser/scripts/browser-doctor.mjs +41 -0
  88. package/packages/omo-codex/plugin/skills/browser/scripts/browser-env.mjs +41 -0
  89. package/packages/omo-codex/plugin/skills/browser/scripts/browser-install.mjs +51 -0
  90. package/packages/omo-codex/plugin/skills/lcx-contribute-bug-fix/SKILL.md +1 -1
  91. package/packages/omo-codex/plugin/skills/lcx-doctor/SKILL.md +3 -3
  92. package/packages/omo-codex/plugin/skills/lcx-report-bug/SKILL.md +5 -5
  93. package/packages/omo-codex/plugin/test/sync-skills-test-support.mjs +1 -0
  94. package/packages/omo-codex/scripts/install-dist/install-local.mjs +90 -17
  95. package/packages/shared-skills/skills/browser/ATTRIBUTION.md +25 -0
  96. package/packages/shared-skills/skills/browser/SKILL.md +89 -0
  97. package/packages/shared-skills/skills/browser/references/commands.md +90 -0
  98. package/packages/shared-skills/skills/browser/references/install.md +71 -0
  99. package/packages/shared-skills/skills/browser/references/owned-engine/README.md +42 -0
  100. package/packages/shared-skills/skills/browser/references/owned-engine/frames-and-humans.md +39 -0
  101. package/packages/shared-skills/skills/browser/references/owned-engine/ladder.md +59 -0
  102. package/packages/shared-skills/skills/browser/references/owned-engine/network.md +42 -0
  103. package/packages/shared-skills/skills/browser/references/recipes/1password.md +59 -0
  104. package/packages/shared-skills/skills/browser/references/remote.md +24 -0
  105. package/packages/shared-skills/skills/browser/scripts/browser-doctor.mjs +41 -0
  106. package/packages/shared-skills/skills/browser/scripts/browser-env.mjs +41 -0
  107. package/packages/shared-skills/skills/browser/scripts/browser-install.mjs +51 -0
  108. package/dist/cli/install-senpi/index.d.ts +0 -1
  109. package/dist/cli/senpi-edition-hint.d.ts +0 -10
  110. package/dist/cli/senpi-platform-flag.d.ts +0 -4
@@ -0,0 +1,89 @@
1
+ ---
2
+ name: browser
3
+ description: "Drives a real browser: sites the user is already signed into, forms and clicks, JS-rendered pages, screenshots, web QA, extension popups, and a human handoff for login, CAPTCHA or OTP. Works through the BrowserSkill extension and its bsk CLI, inside the user's own browser, in a separate Agent Window. Use for any interactive browser task; not for a plain search or an unblocked static fetch."
4
+ ---
5
+
6
+ # Browser
7
+
8
+ Two engines live behind this skill. Choose before you act:
9
+
10
+ | You need | Engine | Where |
11
+ |---|---|---|
12
+ | A site the user is signed into, a form, a click-through, a screenshot, web QA, an extension popup | **attached** — the user's own browser | this file |
13
+ | A throwaway profile, bot-scoring evasion, a CAPTCHA, network interception, a QA flight trace, coordinate control | **owned** — a browser your code launches | [references/owned-engine/README.md](references/owned-engine/README.md) |
14
+ | Text out of a URL, a 403 bypass, a platform that blocks fetchers | neither | the `ultimate-browsing` skill |
15
+
16
+ **Attached is the default,** because it is the only engine carrying the user's logins and the only
17
+ one where a human is a single command away. The owned engine is an opt-in local install, not part
18
+ of this package.
19
+
20
+ ## Step 0 — prove the stack before you drive it
21
+
22
+ ```bash
23
+ node "<skill-root>/scripts/browser-doctor.mjs"
24
+ ```
25
+
26
+ It reports one of four states and what to do next:
27
+
28
+ | State | Meaning | Next |
29
+ |---|---|---|
30
+ | `ready` | CLI, daemon and a connected browser | start a session |
31
+ | `no-extension` | CLI works, no browser is connected | give the user the store link the doctor printed, wait, re-run |
32
+ | `no-cli` | `bsk` is not installed | `node "<skill-root>/scripts/browser-install.mjs"`, then re-run |
33
+ | `no-browser-support` | the platform has no supported browser | say so and stop |
34
+
35
+ **Never substitute another browser for a missing one.** A headless browser you launch yourself has
36
+ none of the user's sessions, so every login turns into a ladder you should not be climbing. If the
37
+ attached engine is unavailable, say which state you hit and ask the user.
38
+
39
+ ## The loop
40
+
41
+ ```bash
42
+ bsk session start --json --no-focus --name "<task>" # keep session_id
43
+ bsk navigate https://example.com --session <id> --json
44
+ bsk observe --session <id> # @eN refs live here
45
+ bsk click @e1 --session <id> --json
46
+ bsk screenshot --session <id> --out ./shot.png --json
47
+ bsk session stop <id> --json # success AND failure
48
+ ```
49
+
50
+ 1. **Observe before every action.** `observe` returns a semantic tree whose `@eN` refs are reissued
51
+ on each call. Read a ref and use it in the same cycle; a ref from two calls ago points somewhere
52
+ else now.
53
+ 2. **Navigation and large DOM changes stale every ref.** Observe again rather than reusing.
54
+ 3. **Two identical failures mean change approach, not retry.** A third identical attempt is a defect.
55
+ 4. **Borrow a user tab explicitly** (`tab list --scope user`, `tab borrow <id>`, `tab return <id>`).
56
+ Borrowing prompts the user; never invent tab ids and never repeat a denied borrow.
57
+ 5. **Always stop the session,** on success and on failure. Stopping also returns borrowed tabs.
58
+
59
+ Command semantics, the flags that behave differently than they read, and the failure table are in
60
+ [references/commands.md](references/commands.md).
61
+
62
+ ## When a human is the only way through
63
+
64
+ Login, CAPTCHA, OTP, a payment confirmation, a consent dialog:
65
+
66
+ ```bash
67
+ bsk request-help --session <id> --prompt "<what you need done>" [--target @eN]
68
+ ```
69
+
70
+ Then observe again. Respect a `cancelled` or `timed_out` answer; do not work around it by changing
71
+ the extension's automation settings.
72
+
73
+ ## Rules
74
+
75
+ - **Never read credentials through the page.** No `evaluate` that extracts a password, token, cookie
76
+ or recovery code. The value of this engine is that the browser is already signed in.
77
+ - **Never clear cookies, cache or site data.** It is the user's real profile; clearing it logs them
78
+ out everywhere. No flow here needs it.
79
+ - **`--no-focus` by default.** The browser belongs to someone who is probably using it.
80
+ - **One short, named session per task,** always stopped.
81
+ - Do not toggle the extension's automation settings, and do not restart the browser to fix a state.
82
+
83
+ ## More
84
+
85
+ - [references/install.md](references/install.md) — installing the CLI and the extension, per OS
86
+ - [references/commands.md](references/commands.md) — command semantics, refs, failure table
87
+ - [references/remote.md](references/remote.md) — agent on one machine, browser on another
88
+ - [references/recipes/1password.md](references/recipes/1password.md) — reading a vault the user has unlocked
89
+ - [references/owned-engine/README.md](references/owned-engine/README.md) — the code-driven engine
@@ -0,0 +1,90 @@
1
+ # Command semantics
2
+
3
+ Everything here cost a failed attempt to learn. Read it before improvising.
4
+
5
+ ## Sessions
6
+
7
+ ```bash
8
+ bsk session start --json --no-focus --name "<task>" # keep session_id
9
+ bsk session list
10
+ bsk session stop <id> --json # positional id, not --session
11
+ ```
12
+
13
+ Every session-scoped command needs `--session <id>`. `--no-focus` keeps the Agent Window from
14
+ stealing focus; drop it only when the user is watching on purpose.
15
+
16
+ ## Reading
17
+
18
+ | Command | Returns |
19
+ |---|---|
20
+ | `observe --session <id>` | semantic tree with `@eN` refs and perception probes — **the default read** |
21
+ | `snapshot --session <id>` | static accessibility tree |
22
+ | `get-html --session <id>` | exact markup, hidden metadata |
23
+ | `screenshot --session <id> --out <path>` | PNG; add `--full-page` for a long capture |
24
+ | `evaluate "<js>" --session <id> --json` | `{ok, value}` — **check `.ok`; exit code 0 does not mean the script succeeded** |
25
+ | `console` / `network` | buffered log lines and responses |
26
+
27
+ **Refs are reissued by every `observe` and `snapshot`.** Read a ref and act on it in the same
28
+ cycle. A ref captured two calls ago silently addresses a different element — this is how a click
29
+ lands on the neighbouring row.
30
+
31
+ `observe --max-tokens <n>` bounds a large page. There is no default cap.
32
+
33
+ ## Acting
34
+
35
+ | Need | Command |
36
+ |---|---|
37
+ | Click | `click @e3 --session <id>` |
38
+ | Fill | `fill @e3 --value "text" --session <id>` |
39
+ | Select | `select @e3 --value "<option value>" --session <id>` |
40
+ | Key | `press Enter --ref @e3 --session <id>` |
41
+ | Hover | `hover @e3 --session <id>` |
42
+ | Scroll into view | `scroll-to @e3 --session <id>` |
43
+ | Wheel | `wheel --delta-y 600 --session <id>` |
44
+ | Focus / blur | `focus @e3` / `blur @e3` |
45
+ | Upload / download | `upload --file <path>` / `download --out <path>` |
46
+
47
+ Traps:
48
+
49
+ - `select` takes the option's **value attribute**, not its visible label.
50
+ - `fill` and `click` accept a CSS selector; **`press --ref` does not** — it answers
51
+ `ref_not_found` for a selector. `press` without `--ref` goes to the focused node.
52
+ - A menu that a `click` opens can be toggled shut by that same click. `focus` then
53
+ `press Enter` opens it reliably.
54
+ - Hover-only controls report `element not visible`: hover the trigger, observe, then act on the
55
+ revealed item's fresh ref. Markers like `[has-submenu]` and `[expanded]` identify triggers;
56
+ `observe --probe-hover` finds one when no marker does, at the cost of touching the live page.
57
+ - The clipboard is unavailable in a window started `--no-focus` (no document focus), so read
58
+ values out of the DOM instead.
59
+
60
+ ## Borrowing a user tab
61
+
62
+ ```bash
63
+ bsk tab list --scope user
64
+ bsk tab borrow <tab-id>
65
+ bsk tab return <tab-id>
66
+ ```
67
+
68
+ Borrowing asks the user to confirm. Never invent a tab id, never repeat a denied borrow, and
69
+ always return what you borrowed (`session stop` also returns them).
70
+
71
+ ## Failures
72
+
73
+ | Symptom | Meaning | Response |
74
+ |---|---|---|
75
+ | `browsers: []` | extension not connected | ask the user to open the browser / enable the extension |
76
+ | daemon missing | idle exit or reboot | none; the next call restarts it |
77
+ | `cdp_failed: Cannot access a chrome-extension:// URL of different extension` | another extension injected a frame, so the debugger cannot attach to that tab | transient and page-specific; collapse the work into one `evaluate` and retry, or navigate away and back |
78
+ | `permission_denied: element not visible` | hover-gated or clipped control | drive its menu instead |
79
+ | `ref_not_found` | stale ref, or a selector passed to `press` | observe again; use a real ref |
80
+ | version skew warning | CLI and extension disagree | `bsk update` after finishing sessions; the extension updates through its store |
81
+
82
+ **Two identical failures select a different approach. A third identical attempt is a defect.**
83
+
84
+ ## Sandboxed hosts
85
+
86
+ If the host kills background children after each command, the daemon cannot survive between calls.
87
+ Set `BSK_AUTO_START=0`, share one `BSK_HOME` across every call, and start
88
+ `bsk daemon start --foreground` in the host's persistent background task. Check readiness with
89
+ `bsk status --json` in a separate call before continuing. Do not loop on launches, delete runtime
90
+ files, or restart a daemon another task is using.
@@ -0,0 +1,71 @@
1
+ # Installing the attached engine
2
+
3
+ Three pieces. The agent can install exactly one of them.
4
+
5
+ | Piece | Channel | Automatable |
6
+ |---|---|---|
7
+ | `bsk` CLI + daemon | upstream installer, GitHub Releases | **yes** |
8
+ | Browser extension | Chrome Web Store / Edge Add-ons | **no — a human clicks Install** |
9
+ | Daemon process | auto-starts on any `bsk` call | nothing to do |
10
+
11
+ ## Supported
12
+
13
+ | | |
14
+ |---|---|
15
+ | Operating systems | macOS (Apple Silicon and Intel), Linux (x64, ARM64), Windows x64 |
16
+ | Browsers | Chrome, Microsoft Edge. Other Chromium browsers work where they accept store builds. |
17
+
18
+ ## 1. The CLI
19
+
20
+ ```bash
21
+ node "<skill-root>/scripts/browser-install.mjs"
22
+ ```
23
+
24
+ That wraps upstream's official installer:
25
+
26
+ ```bash
27
+ # macOS / Linux
28
+ curl -fsSL https://raw.githubusercontent.com/Tencent/BrowserSkill/main/install.sh | sh
29
+ export PATH="${BSK_INSTALL_DIR:-$HOME/.local/bin}:$PATH"
30
+
31
+ # Windows (PowerShell)
32
+ irm https://raw.githubusercontent.com/Tencent/BrowserSkill/main/install.ps1 | iex
33
+ ```
34
+
35
+ The Unix installer cannot change its parent shell's `PATH`. A shell opened before the install may
36
+ still miss it, so either re-export in each call or use the absolute path (`$HOME/.local/bin/bsk`,
37
+ `bsk.exe` on Windows). Verify with `bsk --version`.
38
+
39
+ ## 2. The extension
40
+
41
+ Give the user the listing for their browser and wait:
42
+
43
+ - Chrome: https://chromewebstore.google.com/detail/hhcmgoofomhgciiibhipgmgkgnoenaoi
44
+ - Edge: https://microsoftedge.microsoft.com/addons/detail/browserskill/emacgiaaaiojkkpkddmmdfhmokgmnikg
45
+
46
+ Then confirm it landed:
47
+
48
+ ```bash
49
+ bsk status --json # a non-empty "browsers" array means connected
50
+ ```
51
+
52
+ An empty `browsers` list is not a failure to work around. Either the browser is closed or the
53
+ extension is not enabled; say which and ask.
54
+
55
+ ## 3. Do not install upstream's skill
56
+
57
+ Upstream ships its own agent skill through `bsk install-skill`. **Do not run it here.** This
58
+ package already provides the skill, and upstream's copy installs under a different name into the
59
+ user skill directory, which takes precedence over a shipped skill — two descriptions of the same
60
+ CLI, one of them shadowing the one that knows about this package's recipes and helper scripts.
61
+
62
+ ## Diagnosing
63
+
64
+ ```bash
65
+ node "<skill-root>/scripts/browser-doctor.mjs" # this package's view
66
+ bsk doctor # upstream's own checks
67
+ bsk logs # daemon log
68
+ ```
69
+
70
+ The daemon exits about ten minutes after the last browser disconnects and is not a system service,
71
+ so a reboot also stops it. Both heal on the next `bsk` call — no repair needed.
@@ -0,0 +1,42 @@
1
+ # The owned engine
2
+
3
+ A browser **your code launches and owns**, driven over CDP, with its own profile. The opposite of
4
+ the attached engine: no user logins, full control.
5
+
6
+ **This package ships no such engine.** These documents describe the contract and the technique, so
7
+ that a locally installed CDP library — or a script you write against one — is driven correctly.
8
+ If no owned engine is installed, say so and stay on the attached engine.
9
+
10
+ ## When it is the right engine
11
+
12
+ | Reason | Why the attached engine cannot |
13
+ |---|---|
14
+ | A throwaway or pinned synthetic profile | the attached engine is the user's real profile |
15
+ | Bot-scoring or fingerprint evasion | you do not get to configure the user's browser |
16
+ | Solving a challenge widget programmatically | needs coordinate control and OCR |
17
+ | Reading the network instead of the DOM | needs request interception on your own target |
18
+ | A QA flight trace (steps, HAR, screenshots) | recording someone's real session is not acceptable |
19
+
20
+ For anything that needs the user's login, the attached engine wins. For extracting text from a
21
+ blocked URL, neither: use the `ultimate-browsing` skill.
22
+
23
+ ## What an owned engine must give you
24
+
25
+ 1. A transport that opens no listening port (launch over a pipe), or an explicit local CDP endpoint.
26
+ 2. An accessibility snapshot with stable in-page refs, and a **compact** form that drops the ref
27
+ map before the tree reaches a model — on a real page that map is roughly half the bytes and the
28
+ client never reads it.
29
+ 3. Locators that resolve those refs in-page, including refs inside cross-origin frames.
30
+ 4. Coordinate input, for what locators cannot address.
31
+ 5. A dialog policy, so `alert` / `confirm` / `beforeunload` can never block a run.
32
+
33
+ ## Reference
34
+
35
+ - [ladder.md](ladder.md) — the escalation ladder, viewport pinning, coordinate control, challenge widgets
36
+ - [network.md](network.md) — read the network instead of the DOM; traces and request interception
37
+ - [frames-and-humans.md](frames-and-humans.md) — cross-origin frames, overlays, dialogs, human handoff
38
+
39
+ ## Cleanup is paired
40
+
41
+ Close the browser and remove the profile directory in the same `finally`. A profile left behind is
42
+ a logged-in browser nobody is watching.
@@ -0,0 +1,39 @@
1
+ # Frames, overlays, dialogs, and the human
2
+
3
+ ## Cross-origin frames and shadow DOM
4
+
5
+ A cross-origin iframe is a separate target with its own snapshot. Reconcile the child trees into
6
+ the parent so one tree describes the whole page, and address child elements through the page-level
7
+ ref — never by fetching a frame handle first. Shadow roots are traversed by the snapshot engine;
8
+ they are not a special case for you.
9
+
10
+ ## Overlays that swallow clicks
11
+
12
+ Two identical misses on an element that is clearly visible usually means something invisible sits
13
+ on top: a consent banner, a modal backdrop, a sticky header, a full-viewport tracking layer.
14
+ Describe the layers at the click point before you retry. Either the overlay is named — dismiss it
15
+ and continue — or nothing is reported, which means the miss has another cause and you climb the
16
+ ladder instead.
17
+
18
+ ## Dialogs never block
19
+
20
+ `alert`, `confirm`, `prompt`, and `beforeunload` must be answered by a policy set when the
21
+ browser is created, not by a listener you hope registers in time. Default to accepting
22
+ (`confirm` true, `prompt` empty). A run that hangs on an unhandled dialog looks exactly like a
23
+ hang with no cause, which is the most expensive kind to diagnose.
24
+
25
+ ## Device emulation
26
+
27
+ Emulating a device is viewport plus user agent plus touch, applied together. Changing only the
28
+ viewport produces a desktop page at a phone size, which is not what you are testing. Re-read the
29
+ viewport after emulating, and re-pin coordinates.
30
+
31
+ ## Handing the page to a human
32
+
33
+ When every rung fails on a login, a one-time code, or a challenge you cannot clear, the human is
34
+ the fallback, not the failure. Bring the tab to the front, state plainly what needs doing, and
35
+ wait on a **condition** — the URL changed, the element appeared — with a bounded timeout, rather
36
+ than on a fixed delay.
37
+
38
+ Respect the answer. A cancelled or timed-out handoff is a stop, and the correct response is to
39
+ report which rung failed with what evidence. Working around the user's refusal is never correct.
@@ -0,0 +1,59 @@
1
+ # The escalation ladder
2
+
3
+ **Two identical failures on the same target mean climb, not retry. A third identical attempt is a
4
+ defect.** The failure itself selects the next rung; never pause to ask which.
5
+
6
+ | Rung | Use | Climb when |
7
+ |---|---|---|
8
+ | 1. snapshot + locator(ref) | anything with a usable ref | the ref is absent, stale, obscured, or the click lands on the wrong node twice |
9
+ | 1b. layer description | a blocking overlay explains two identical misses | no overlay is reported, or it is gone and the click still misses |
10
+ | 2. coordinates | canvas, extension popups, custom controls, drag surfaces | the click misses, or the screenshot and the coordinates disagree |
11
+ | 3. pin the viewport, redo rung 2 | coordinate drift after a resize, a DPI change, or a foreign tab | coordinates land correctly but the widget still refuses input |
12
+ | 4. the challenge widget | a challenge is the blocker | it clears and the flow still stalls |
13
+ | 5. read the browser log | a browser-level failure | the log names a cause outside the page |
14
+
15
+ **Rung 5 ends in a written diagnosis, never a speculative code change.** A missing entitlement, a
16
+ dead extension service worker, an unavailable authenticator — none of those is fixed by editing
17
+ automation code.
18
+
19
+ ## Pin the viewport before trusting any coordinate
20
+
21
+ Coordinate input acts in viewport pixels. When the render surface and the coordinate space
22
+ disagree, every coordinate is off by the same constant and retrying just repeats the miss. Set the
23
+ device metrics explicitly (width, height, `deviceScaleFactor`, and a matching viewport), re-read
24
+ the page's viewport size, then take a **fresh** screenshot. Coordinates read off an unpinned
25
+ screenshot are stale. Pin every page you act on, including tabs you did not open.
26
+
27
+ ## Refs
28
+
29
+ - Refs die on every new snapshot. Pass a ref straight from the latest snapshot; never reuse one
30
+ across snapshots and never put one in a CSS selector.
31
+ - Refs stay page-level. A child-frame ref still resolves through the page, so you never fetch a
32
+ frame handle to click inside an iframe.
33
+ - Always compact a snapshot before sending it to a model.
34
+
35
+ ## Challenge widgets
36
+
37
+ A challenge is an obstacle on the path, not a stop sign: clear it, confirm it cleared, continue.
38
+ Three shapes, three techniques:
39
+
40
+ - **Checkbox** (the common embedded widgets): the control lives in a nested frame, so address it
41
+ by viewport coordinates.
42
+ - **Slider / puzzle**: drag along a path with enough intermediate steps that the motion reads as
43
+ human. Raise the step count before you change the endpoints.
44
+ - **Text or number**: OCR the region. On macOS the system vision framework is the cheapest
45
+ accurate option; elsewhere pass your own OCR function or a vision model.
46
+ - **Image grid**: annotate a screenshot, then click per cell.
47
+
48
+ **Bounds measured against an unpinned viewport are wrong by a constant offset.** If clicks have
49
+ been landing wrong, drop to rung 3 first, then re-read the bounds.
50
+
51
+ Verify with a fresh snapshot that the challenge is gone. Verification is part of the action, not a
52
+ separate optimistic assumption.
53
+
54
+ ## Delegating the pixel loop
55
+
56
+ Coordinate work and challenge solving are iterative: screenshot, reason, act, verify. When that
57
+ loop would consume your own context, delegate the blocked page to a subagent armed with these
58
+ references and have it return the post-solve state. Drive it yourself when the flow is short or
59
+ the state is already in your hands.
@@ -0,0 +1,42 @@
1
+ # Read the network, not the DOM
2
+
3
+ When a page renders a list, a table, or search results, the data almost always arrived as JSON one
4
+ request earlier. That JSON is complete, typed, free of markup, and immune to the layout changing
5
+ next week. Watching traffic costs nothing on a target you own, and the page cannot observe it.
6
+
7
+ **Snoop before you scrape.** Scroll-and-parse is the fallback, not the default.
8
+
9
+ ## Wait for a request, never for a clock
10
+
11
+ A fixed sleep is a guess that fails on a slower machine and wastes time on a faster one. Subscribe
12
+ to the response you expect **before** triggering the action, then await that signal with a bounded
13
+ timeout. This is the same rule that governs test code, for the same reason.
14
+
15
+ ## Collecting through infinite scroll
16
+
17
+ Drive scroll and extraction as a generator: scroll, collect what is new, stop on a target count or
18
+ a scroll ceiling. Bound both, so a page that keeps producing cannot run forever.
19
+
20
+ ## Flight traces
21
+
22
+ For QA evidence, record a trace: one entry per step with before/after screenshots, the network
23
+ log, and console output, written as a line-delimited log plus an HAR. That triple is what makes a
24
+ failure reconstructable afterwards instead of re-runnable-in-theory.
25
+
26
+ Cap recorded body sizes. An untrimmed trace of a media-heavy page is mostly bytes nobody reads.
27
+
28
+ ## Request interception
29
+
30
+ Route matching by glob, regular expression, or predicate lets you stub an endpoint, fail one
31
+ request to test an error path, or block third-party noise. Enable interception on the first route
32
+ and disable it on dispose — leaving it on slows every later navigation.
33
+
34
+ ## Cookies
35
+
36
+ Injecting a cookie export into your own profile is legitimate for ordinary sessions and useless
37
+ for the ones you most want: accounts whose risk engines bind a session to a device will invalidate
38
+ it, and a password manager's session is tab-bound and never portable. Sanitize what you do inject
39
+ (drop expired entries, keep host-prefixed cookies secure and root-scoped).
40
+
41
+ **Never copy a session out of the user's real browser profile.** If you need their login, that is
42
+ the attached engine's job.
@@ -0,0 +1,59 @@
1
+ # Reading a 1Password vault the user has unlocked
2
+
3
+ Scope: reading an item out of a vault **the user has already unlocked in their browser**. Signing
4
+ in is the user's job; see "Not in scope" below.
5
+
6
+ ## The fact that shapes everything
7
+
8
+ 1Password web is zero-knowledge. The unlocked vault key lives in **one tab's `sessionStorage`**,
9
+ not in a cookie. So:
10
+
11
+ - A **new** tab pointed at the vault redirects to the sign-in page even while a signed-in tab sits
12
+ right beside it. Opening your own tab does not work and never will.
13
+ - Navigating the authenticated tab away, or closing it, **destroys the session** until a human
14
+ signs in again.
15
+
16
+ Therefore: borrow the existing tab, read, return it. Do not open, do not navigate away, do not
17
+ "refresh to fix" anything.
18
+
19
+ ```bash
20
+ bsk tab list --scope user # find the already-signed-in tab
21
+ bsk tab borrow <tab-id>
22
+ # ... read ...
23
+ bsk tab return <tab-id>
24
+ ```
25
+
26
+ ## Check liveness by URL, never by title
27
+
28
+ The app mutates its URL hash without updating the document title, so a signed-in vault can still
29
+ report a sign-in title. A live session's `location.href` contains the app path and never the
30
+ sign-in path. Read the URL.
31
+
32
+ ## The browser may be shared
33
+
34
+ Other automation may be driving the same window.
35
+
36
+ - **Clear the search box before typing.** It often holds someone else's query, and typing alone
37
+ appends to it, returning confident results for the wrong search.
38
+ - **Read the autocomplete listbox, not the results grid.** The grid holds whatever query was last
39
+ committed, possibly by another actor. Scraping every option on the page silently merges two
40
+ different searches.
41
+ - **Leave the tab on the item list** when you are done, so the next actor starts clean.
42
+
43
+ ## If it is locked
44
+
45
+ ```bash
46
+ bsk request-help --session <id> --prompt "Please unlock 1Password in this tab, then continue."
47
+ ```
48
+
49
+ Then observe again.
50
+
51
+ ## Not in scope
52
+
53
+ - **Do not type a master password, and do not automate the sign-in form.** Only the account owner
54
+ signs in. If the vault is locked, hand it to them.
55
+ - **Do not extract secrets through the page** beyond the single value the user asked for, and
56
+ report it masked (`abcd…9fb0 (len 37)`) unless they asked for the exact string.
57
+ - **Do not copy, inject, or reuse session cookies.** The session is device- and tab-bound; cookie
58
+ reuse cannot work and trips account protections.
59
+ - Clear the clipboard if you ever populate it.
@@ -0,0 +1,24 @@
1
+ # Agent here, browser there
2
+
3
+ The agent and the browser do not have to be the same machine. Upstream supports this directly;
4
+ nothing in this package needs to bridge it.
5
+
6
+ Shape:
7
+
8
+ - The **browser machine** needs only the extension.
9
+ - The **agent machine** runs the `bsk` CLI and this skill.
10
+ - They are joined by a daemon in server mode plus a pairing link.
11
+
12
+ ```bash
13
+ bsk daemon start --mode server # on the machine the agent runs on
14
+ ```
15
+
16
+ Then follow upstream's pairing guide, which owns the current flag surface and the TLS
17
+ prerequisites: https://github.com/Tencent/BrowserSkill/blob/main/docs/remote-extension-connection.md
18
+
19
+ Once paired, every command in `commands.md` behaves identically; `--out` paths still resolve on
20
+ the machine that ran the command.
21
+
22
+ If pairing is not configured, say so and ask. Do not substitute a browser on the agent machine:
23
+ the point of the attached engine is the sessions that live in the user's browser, and a local
24
+ browser has none of them.
@@ -0,0 +1,41 @@
1
+ #!/usr/bin/env node
2
+ import { platform } from "node:os"
3
+ import { connectedBrowsers, readStatus, resolveCli, STORE_LISTINGS, SUPPORTED_PLATFORMS } from "./browser-env.mjs"
4
+
5
+ const REMEDIES = {
6
+ ready: "Start a session: bsk session start --json --no-focus --name \"<task>\"",
7
+ "no-extension": `Ask the user to install and enable the extension, then re-run this doctor.\n Chrome: ${STORE_LISTINGS.chrome}\n Edge: ${STORE_LISTINGS.edge}`,
8
+ "no-cli": "Install the CLI: node \"<skill-root>/scripts/browser-install.mjs\", then re-run this doctor.",
9
+ "no-browser-support": "This platform has no supported browser. Say so and stop; do not substitute another engine.",
10
+ }
11
+
12
+ async function diagnose() {
13
+ if (!SUPPORTED_PLATFORMS.has(platform())) {
14
+ return { state: "no-browser-support", platform: platform() }
15
+ }
16
+ const cli = await resolveCli()
17
+ if (cli === undefined) return { state: "no-cli", platform: platform() }
18
+
19
+ let status
20
+ try {
21
+ status = await readStatus(cli)
22
+ } catch (error) {
23
+ return { state: "no-extension", platform: platform(), cli, detail: `status failed: ${error.message}` }
24
+ }
25
+ const browsers = connectedBrowsers(status)
26
+ if (browsers.length === 0) return { state: "no-extension", platform: platform(), cli }
27
+ return { state: "ready", platform: platform(), cli, browsers: browsers.length }
28
+ }
29
+
30
+ const report = await diagnose()
31
+ const json = process.argv.includes("--json")
32
+ if (json) {
33
+ console.log(JSON.stringify({ ...report, remedy: REMEDIES[report.state] }, null, 2))
34
+ } else {
35
+ console.log(`state: ${report.state}`)
36
+ if (report.cli) console.log(`cli: ${report.cli}`)
37
+ if (report.browsers) console.log(`browsers connected: ${report.browsers}`)
38
+ if (report.detail) console.log(`detail: ${report.detail}`)
39
+ console.log(`\n${REMEDIES[report.state]}`)
40
+ }
41
+ process.exit(report.state === "ready" ? 0 : 1)
@@ -0,0 +1,41 @@
1
+ import { execFile } from "node:child_process"
2
+ import { existsSync } from "node:fs"
3
+ import { homedir, platform } from "node:os"
4
+ import { join } from "node:path"
5
+ import { promisify } from "node:util"
6
+
7
+ const run = promisify(execFile)
8
+
9
+ export const STORE_LISTINGS = {
10
+ chrome: "https://chromewebstore.google.com/detail/hhcmgoofomhgciiibhipgmgkgnoenaoi",
11
+ edge: "https://microsoftedge.microsoft.com/addons/detail/browserskill/emacgiaaaiojkkpkddmmdfhmokgmnikg",
12
+ }
13
+
14
+ export const SUPPORTED_PLATFORMS = new Set(["darwin", "linux", "win32"])
15
+
16
+ export function localBinCandidates() {
17
+ const bin = join(homedir(), ".local", "bin")
18
+ return platform() === "win32" ? [join(bin, "bsk.exe"), join(bin, "bsk")] : [join(bin, "bsk")]
19
+ }
20
+
21
+ export async function resolveCli() {
22
+ for (const candidate of localBinCandidates()) {
23
+ if (existsSync(candidate)) return candidate
24
+ }
25
+ try {
26
+ await run("bsk", ["--version"], { timeout: 15000, windowsHide: true })
27
+ return "bsk"
28
+ } catch {
29
+ return undefined
30
+ }
31
+ }
32
+
33
+ export async function readStatus(cli) {
34
+ const { stdout } = await run(cli, ["status", "--json"], { timeout: 30000, windowsHide: true })
35
+ return JSON.parse(stdout)
36
+ }
37
+
38
+ export function connectedBrowsers(status) {
39
+ const browsers = status?.browsers
40
+ return Array.isArray(browsers) ? browsers : []
41
+ }
@@ -0,0 +1,51 @@
1
+ #!/usr/bin/env node
2
+ import { spawn } from "node:child_process"
3
+ import { platform } from "node:os"
4
+ import { resolveCli, STORE_LISTINGS, SUPPORTED_PLATFORMS } from "./browser-env.mjs"
5
+
6
+ const UNIX_INSTALL = "curl -fsSL https://raw.githubusercontent.com/Tencent/BrowserSkill/main/install.sh | sh"
7
+ const WINDOWS_INSTALL = "irm https://raw.githubusercontent.com/Tencent/BrowserSkill/main/install.ps1 | iex"
8
+
9
+ function installCommand() {
10
+ return platform() === "win32"
11
+ ? { file: "powershell", args: ["-NoProfile", "-Command", WINDOWS_INSTALL], display: WINDOWS_INSTALL }
12
+ : { file: "sh", args: ["-c", UNIX_INSTALL], display: UNIX_INSTALL }
13
+ }
14
+
15
+ function spawnInstaller({ file, args }) {
16
+ return new Promise((resolve) => {
17
+ const child = spawn(file, args, { stdio: "inherit", windowsHide: true })
18
+ child.on("error", () => resolve(1))
19
+ child.on("close", (code) => resolve(code ?? 1))
20
+ })
21
+ }
22
+
23
+ if (!SUPPORTED_PLATFORMS.has(platform())) {
24
+ console.error(`unsupported platform: ${platform()}. macOS, Linux and Windows x64 only.`)
25
+ process.exit(1)
26
+ }
27
+
28
+ if (await resolveCli()) {
29
+ console.log("bsk is already installed; run browser-doctor.mjs to check the extension.")
30
+ process.exit(0)
31
+ }
32
+
33
+ const command = installCommand()
34
+ console.log(`installing the bsk CLI with the upstream installer:\n ${command.display}\n`)
35
+ const code = await spawnInstaller(command)
36
+ if (code !== 0) {
37
+ console.error(`\ninstaller exited ${code}. Report this instead of retrying blindly.`)
38
+ process.exit(code)
39
+ }
40
+
41
+ console.log([
42
+ "",
43
+ "CLI installed. Two things remain, and only the user can do the first:",
44
+ "",
45
+ "1. Install the browser extension from the store, then enable it:",
46
+ ` Chrome: ${STORE_LISTINGS.chrome}`,
47
+ ` Edge: ${STORE_LISTINGS.edge}`,
48
+ "2. Re-run: node \"<skill-root>/scripts/browser-doctor.mjs\"",
49
+ "",
50
+ "If PATH does not pick up the CLI in this shell, use its absolute path under ~/.local/bin.",
51
+ ].join("\n"))
@@ -1 +0,0 @@
1
- export * from "@oh-my-opencode/omo-senpi/install";