@sous-io/sous 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +154 -0
- package/bin/run.js +17 -0
- package/bin/xcv +5 -0
- package/package.json +81 -0
- package/shared-prompts/_partials/resume-task.md +51 -0
- package/shared-prompts/_partials/sub-agent-delegation.md +32 -0
- package/shared-prompts/_partials/update-task-file.md +52 -0
- package/shared-prompts/memories/automated-browser-tasks/INDEX.tpl.md +52 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/SKILL.tpl.md +102 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/examples/auth-failure-handling.mjs +81 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/examples/chained-workflow.mjs +126 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/examples/simple-fetch.mjs +92 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/references/architecture.md +61 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/references/auth-and-sessions.md +65 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/references/ctx-api.md +96 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/references/installation.md +104 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/references/script-conventions.md +243 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/chrome-state.mjs +148 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/debug.mjs +383 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/debug.spec.mjs +267 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/eslint.config.mjs +56 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/harness.mjs +169 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/keyring.mjs +59 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/logger.mjs +25 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/params.mjs +140 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/run.mjs +140 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/settings.tpl.mjs +1 -0
- package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/utils.mjs +185 -0
- package/shared-prompts/skills/automated-browser-tasks/create-automated-browser-task/SKILL.tpl.md +52 -0
- package/shared-prompts/skills/automated-browser-tasks/running-automated-browser-tasks/SKILL.tpl.md +59 -0
- package/shared-prompts/skills/automated-browser-tasks/update-automated-browser-task/SKILL.tpl.md +47 -0
- package/shared-prompts/skills/control-flow/approve/SKILL.tpl.md +26 -0
- package/shared-prompts/skills/control-flow/opine/SKILL.tpl.md +58 -0
- package/shared-prompts/skills/control-flow/repeat/SKILL.tpl.md +27 -0
- package/shared-prompts/skills/control-flow/research/SKILL.tpl.md +34 -0
- package/shared-prompts/skills/sous-skills/about-agent-skills/SKILL.tpl.md +177 -0
- package/shared-prompts/skills/sous-skills/about-agent-skills/examples/about-something.md +45 -0
- package/shared-prompts/skills/sous-skills/about-agent-skills/examples/do-something.md +33 -0
- package/shared-prompts/skills/sous-skills/about-agent-skills/references/advanced-patterns.md +87 -0
- package/shared-prompts/skills/sous-skills/about-agent-skills/references/commands.md +46 -0
- package/shared-prompts/skills/sous-skills/about-agent-skills/references/frontmatter.md +25 -0
- package/shared-prompts/skills/sous-skills/about-agent-skills/references/substitutions.md +50 -0
- package/shared-prompts/skills/sous-skills/about-liquid-templates/SKILL.tpl.md +268 -0
- package/shared-prompts/skills/sous-skills/about-liquid-templates/references/liquid-filters.md +82 -0
- package/shared-prompts/skills/sous-skills/about-sous/SKILL.tpl.md +51 -0
- package/shared-prompts/skills/sous-skills/create-skill/SKILL.tpl.md +114 -0
- package/shared-prompts/skills/task-files/about-task-files/SKILL.tpl.md +122 -0
- package/shared-prompts/skills/task-files/continue-task-in-new-branch/SKILL.tpl.md +80 -0
- package/shared-prompts/skills/task-files/go/SKILL.tpl.md +14 -0
- package/shared-prompts/skills/task-files/resume-task/SKILL.tpl.md +13 -0
- package/shared-prompts/skills/task-files/start-task/SKILL.tpl.md +93 -0
- package/shared-prompts/skills/task-files/update/SKILL.tpl.md +14 -0
- package/shared-prompts/skills/task-files/update-task-file/SKILL.tpl.md +13 -0
- package/src/base-command.ts +163 -0
- package/src/commands/build.ts +196 -0
- package/src/commands/clear.ts +71 -0
- package/src/commands/compile.ts +95 -0
- package/src/commands/launch.ts +111 -0
- package/src/commands/prune.ts +48 -0
- package/src/lib/build-service.ts +258 -0
- package/src/lib/config-discovery.ts +199 -0
- package/src/lib/env-local.ts +195 -0
- package/src/lib/include-resolver.ts +146 -0
- package/src/lib/markdown-compiler.ts +580 -0
- package/src/lib/pid-service.ts +88 -0
- package/src/lib/settings.ts +695 -0
- package/src/lib/state.ts +135 -0
- package/src/lib/watch-service.ts +115 -0
- package/src/templating/filters/bullet-list.ts +9 -0
- package/src/templating/filters/index.ts +8 -0
- package/src/templating/init-liquid-engine.ts +82 -0
- package/src/templating/lib/glob-files.ts +74 -0
- package/src/templating/lib/import-export.ts +32 -0
- package/src/templating/lib/tag-args.ts +19 -0
- package/src/templating/tags/exportScalarVarsJs.ts +43 -0
- package/src/templating/tags/getFiles.ts +89 -0
- package/src/templating/tags/index.ts +14 -0
- package/src/templating/tags/listFiles.ts +54 -0
- package/src/templating/tags/showVars.ts +22 -0
- package/src/utils/formatting.ts +338 -0
- package/src/utils/prompts.ts +19 -0
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
# Installation
|
|
2
|
+
|
|
3
|
+
## Runtime & platform
|
|
4
|
+
|
|
5
|
+
- Node.js ≥ 22.
|
|
6
|
+
- Linux with a GNOME-keyring-compatible Secret Service (the user's Chrome must
|
|
7
|
+
have stored its Safe Storage key there — true after Chrome has run once on a
|
|
8
|
+
desktop session with an unlocked keyring).
|
|
9
|
+
- Google Chrome installed with at least one profile the user has logged into.
|
|
10
|
+
|
|
11
|
+
macOS (Keychain) and Windows (DPAPI) are not yet supported by `keyring.mjs`.
|
|
12
|
+
|
|
13
|
+
## Dependencies
|
|
14
|
+
|
|
15
|
+
The runner and harness import these at runtime; the framework does not bundle
|
|
16
|
+
them. Install them **at the consuming project's root** — NOT globally.
|
|
17
|
+
|
|
18
|
+
Why not global: the scripts use ESM `import 'playwright'`. ESM resolves bare
|
|
19
|
+
imports by walking *up* the directory tree from the importing file looking for a
|
|
20
|
+
`node_modules`. The compiled runner lives at
|
|
21
|
+
`<projectRoot>/.claude/skills/about-automated-browser-tasks/scripts/run.mjs`, so a
|
|
22
|
+
`node_modules` at `<projectRoot>` is found by walking up; a global npm install is
|
|
23
|
+
never on that resolution path.
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
cd <projectRoot> # the repo root that contains .claude/skills/
|
|
27
|
+
npm install playwright better-sqlite3 dbus-next
|
|
28
|
+
npx playwright install chromium
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Add a `package.json` at `<projectRoot>` if none exists (`{"type":"module","private":true}`)
|
|
32
|
+
and gitignore `node_modules/`.
|
|
33
|
+
|
|
34
|
+
- `playwright` — headless browser automation.
|
|
35
|
+
- `better-sqlite3` — reads Chrome's `Cookies` SQLite DB.
|
|
36
|
+
- `dbus-next` — pure-JS D-Bus client for the keyring (no Python, no native build).
|
|
37
|
+
|
|
38
|
+
Tested with: Playwright 1.61, better-sqlite3 12.x, Node 22, Chrome cookie format
|
|
39
|
+
v11, Ubuntu 22.04.
|
|
40
|
+
|
|
41
|
+
## Project wiring (via sous)
|
|
42
|
+
|
|
43
|
+
A downstream project compiles this bundle into its skills directory and compiles
|
|
44
|
+
`settings.tpl.mjs` → `settings.mjs` (sibling of `run.mjs`) so scripts get
|
|
45
|
+
`ctx.settings`. Example compilation targets:
|
|
46
|
+
|
|
47
|
+
```js
|
|
48
|
+
compilation: {
|
|
49
|
+
targets: [
|
|
50
|
+
{
|
|
51
|
+
// The skills (SKILL.tpl.md, references, examples, scripts) → skills dir
|
|
52
|
+
entryGlob: "${sousRootPath}/shared-prompts/skills/automated-browser-tasks/**/*",
|
|
53
|
+
outputs: [{ destinationDir: "${projectRoot}/.claude/skills" }],
|
|
54
|
+
},
|
|
55
|
+
],
|
|
56
|
+
}
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Define project values (`chromeProfile`, base URLs, resource IDs, …) in `_vars`. The
|
|
60
|
+
`{% exportScalarVarsJs %}` tag in `settings.tpl.mjs` emits all in-scope scalars
|
|
61
|
+
as the runtime settings module — no per-key wiring needed. Be sure to define
|
|
62
|
+
`browserAutomationScriptsDir` (the absolute path to the project's task scripts)
|
|
63
|
+
in `_vars` — both the runtime and the task manifest below rely on it.
|
|
64
|
+
|
|
65
|
+
## Task manifest in core memory
|
|
66
|
+
|
|
67
|
+
So the agent always knows which browser tasks exist (without relying on a skill
|
|
68
|
+
trigger firing), render the shared memory partial into the project's memory source
|
|
69
|
+
tree, then `@include` it from a core-memory file. It renders a live list of every
|
|
70
|
+
task script via `{% getFiles … import="meta" %}`, reading each script's `meta`.
|
|
71
|
+
|
|
72
|
+
`@include` does NOT substitute variables, so you cannot `@`-include the shared
|
|
73
|
+
`INDEX.tpl.md` by an absolute `${...}` path. Instead, add a compilation target that
|
|
74
|
+
renders it into your memory tree (exactly how `runtimeContext` emits
|
|
75
|
+
`session-context.md`):
|
|
76
|
+
|
|
77
|
+
```js
|
|
78
|
+
// A target that renders the shared manifest into the project's memory source.
|
|
79
|
+
const browserTaskManifest = {
|
|
80
|
+
entryPoint: "${sousRootPath}/shared-prompts/memories/automated-browser-tasks/INDEX.tpl.md",
|
|
81
|
+
outputs: [
|
|
82
|
+
{ destinationFile: "${memoryRoot}/tools/automated-browser-tasks.md" },
|
|
83
|
+
],
|
|
84
|
+
};
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
Order this target BEFORE the memories target that composes core memory. Then pull
|
|
88
|
+
it into a memory file (e.g. `tools/README.md`) with a plain relative Sous include:
|
|
89
|
+
put an `@`-prefixed line containing just the rendered filename
|
|
90
|
+
(`automated-browser-tasks.md`) on its own line in that file.
|
|
91
|
+
|
|
92
|
+
The manifest auto-rebuilds on every `xcv build`, so newly created tasks appear
|
|
93
|
+
automatically. It requires `browserAutomationScriptsDir` to be in scope (the
|
|
94
|
+
absolute path to the task scripts).
|
|
95
|
+
|
|
96
|
+
## Verifying
|
|
97
|
+
|
|
98
|
+
Run any example script by absolute path:
|
|
99
|
+
|
|
100
|
+
```bash
|
|
101
|
+
node <scriptsDir>/run.mjs <scriptsDir>/../examples/simple-fetch.mjs --url=https://example.com
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
A clean run prints extracted cookie counts, a browser-ready line, and the result.
|
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
# Script Conventions
|
|
2
|
+
|
|
3
|
+
These rules are non-negotiable. A reviewer (or linter) should be able to reject a
|
|
4
|
+
script that violates them.
|
|
5
|
+
|
|
6
|
+
## Quality bar: bulletproof or it doesn't ship
|
|
7
|
+
|
|
8
|
+
A flaky script is a broken script. "Works most of the time" is failure. Write for
|
|
9
|
+
100% reliability across many consecutive and parallel runs from the first draft —
|
|
10
|
+
do not ship something that "usually works" and plan to harden later.
|
|
11
|
+
|
|
12
|
+
The single greatest source of flakiness is **guessing about timing instead of
|
|
13
|
+
waiting for facts**. Every wait must key off a concrete, observable condition that
|
|
14
|
+
*proves* the thing you need is ready. Spend the extra time to find that signal.
|
|
15
|
+
Arbitrary delays (`waitForTimeout`) are the enemy — see [Waits](#waits); they are
|
|
16
|
+
effectively banned.
|
|
17
|
+
|
|
18
|
+
Before writing a single wait or selector, **observe the real page.** Do not assume
|
|
19
|
+
DOM structure. Build your waits from what you actually see — selectors invented
|
|
20
|
+
from imagination are how you get a script that passes once and fails in CI.
|
|
21
|
+
|
|
22
|
+
`ctx.debug` exists for exactly this (full surface in `ctx-api.md`):
|
|
23
|
+
|
|
24
|
+
- `ctx.debug.dump()` — snapshot URL, title, text, screenshot, and HTML to disk.
|
|
25
|
+
- `ctx.debug.describe(selector)` / `ctx.debug.clickables()` — see whether a
|
|
26
|
+
selector matches and what the real interactive elements are (often a
|
|
27
|
+
click-handled `div`, not the `<a>`/`<button>` you assumed).
|
|
28
|
+
- `ctx.debug.findText(text)` — locate text and the clickable ancestor to target.
|
|
29
|
+
- `ctx.debug.watch(() => metric)` — sample a metric over time to find the *moment*
|
|
30
|
+
the page is genuinely ready (this is how you discover that `networkidle` fired
|
|
31
|
+
on an empty shell), then wait on that concrete signal.
|
|
32
|
+
|
|
33
|
+
On an unexpected throw, the harness auto-captures a failure snapshot
|
|
34
|
+
(`result.debugDir`) — check it first when a run fails. Remove file-writing debug
|
|
35
|
+
calls (`dump`/`screenshot`/`html`) once the script is solid; keep them out of hot
|
|
36
|
+
loops.
|
|
37
|
+
|
|
38
|
+
## Structure
|
|
39
|
+
|
|
40
|
+
`execute(ctx)` is a thin orchestrator that reads like a table of contents. All
|
|
41
|
+
real work lives in small, named step functions defined below `execute` in the
|
|
42
|
+
same file.
|
|
43
|
+
|
|
44
|
+
- One discrete action per step function (navigate, dismiss, extract, parse…).
|
|
45
|
+
- ≤ 30 lines per function; 10 or fewer is ideal.
|
|
46
|
+
- Module-level functions, not class methods.
|
|
47
|
+
- Generic patterns → `ctx.utils`. Site-specific patterns → step functions (which
|
|
48
|
+
a project may later factor into shared libs it imports).
|
|
49
|
+
|
|
50
|
+
## Doc-blocks
|
|
51
|
+
|
|
52
|
+
EVERY function — `execute` included — has a proper JSDoc block: a description
|
|
53
|
+
line plus `@param` for every argument and `@returns`. Use
|
|
54
|
+
`@returns {Promise<void>}` for functions that return nothing. Single-line
|
|
55
|
+
`/** … */` comments are NOT sufficient.
|
|
56
|
+
|
|
57
|
+
## Destructuring
|
|
58
|
+
|
|
59
|
+
Each function destructures the members it needs off `ctx` (and off `params`) at
|
|
60
|
+
the top of its body, so the body never repeats `ctx.`/`params.` prefixes:
|
|
61
|
+
|
|
62
|
+
```js
|
|
63
|
+
async function navigateToRepo(ctx, baseUrl, repoId) {
|
|
64
|
+
const { page, logger, timeout, checkAuth } = ctx;
|
|
65
|
+
...
|
|
66
|
+
}
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
## Params (`meta.params`)
|
|
70
|
+
|
|
71
|
+
The framework resolves and validates params before `execute` runs. Scripts never
|
|
72
|
+
validate their own params. Resolution priority (low → high):
|
|
73
|
+
`ctx.settings` < `meta.params[x].default` < explicit (CLI) params.
|
|
74
|
+
|
|
75
|
+
Each param spec:
|
|
76
|
+
|
|
77
|
+
| Field | Type | Meaning |
|
|
78
|
+
|-------|------|---------|
|
|
79
|
+
| `required` | boolean | Error if nothing resolves. |
|
|
80
|
+
| `default` | any | Fallback value. |
|
|
81
|
+
| `description` | string | Be genuinely descriptive: what it is, where to find it, how it's used, consequence of omitting. Shown in listings and errors. |
|
|
82
|
+
| `validate` | `RegExp` \| `Function` | See below. |
|
|
83
|
+
| `invalidMessage` | string | Error for a failing RegExp, or a `validate` fn returning `false`. |
|
|
84
|
+
|
|
85
|
+
`validate`:
|
|
86
|
+
- **RegExp** — resolved value (as string) must match.
|
|
87
|
+
- **Function** `(value, resolvedParams) => true | false | string` — `true` valid;
|
|
88
|
+
a returned `string` is used as the error; `false` falls back to `invalidMessage`.
|
|
89
|
+
The function gets all resolved params, enabling cross-param checks.
|
|
90
|
+
|
|
91
|
+
All failures across params are collected into one `ParamError`.
|
|
92
|
+
|
|
93
|
+
## Logging
|
|
94
|
+
|
|
95
|
+
Use `ctx.logger`, never `console.log`. Create a child per section:
|
|
96
|
+
`const log = ctx.logger.child('navigate')`. Output is
|
|
97
|
+
`[script-name:section] message`. Levels: `info`, `warn`, `error`.
|
|
98
|
+
|
|
99
|
+
## Return shape
|
|
100
|
+
|
|
101
|
+
Return a plain object. Common keys:
|
|
102
|
+
|
|
103
|
+
- `found: boolean` — whether the target content was located.
|
|
104
|
+
- `content: string` — extracted content (when found).
|
|
105
|
+
- `outputFile: string` — path for the runner to write `content` to.
|
|
106
|
+
- a URL key (e.g. `buildUrl`) — where content was found.
|
|
107
|
+
- `message: string` — human-readable explanation, especially on failure.
|
|
108
|
+
|
|
109
|
+
On `found: false`, include diagnostics (`pageTextPreview`, `message`).
|
|
110
|
+
|
|
111
|
+
## Verify every action
|
|
112
|
+
|
|
113
|
+
Do not assume an action took effect — prove it. After every navigation or click
|
|
114
|
+
that changes state, wait for a signal that confirms the *intended outcome*:
|
|
115
|
+
|
|
116
|
+
- After a navigation: `await page.waitForURL(/expected-path/)`, or wait for an
|
|
117
|
+
element that only exists on the destination.
|
|
118
|
+
- After a click that should open a view: wait for that view's content, not just
|
|
119
|
+
for the click to return.
|
|
120
|
+
- After triggering content load: wait for the content to be present AND non-empty
|
|
121
|
+
(e.g. a `<pre>` whose text length exceeds a threshold), not merely attached.
|
|
122
|
+
|
|
123
|
+
A click with `{ force: true }` is fire-and-forget: it bypasses Playwright's
|
|
124
|
+
actionability checks (visible, stable, not covered) and reports success even when
|
|
125
|
+
it lands on nothing. Prefer a plain click — Playwright then auto-waits for the
|
|
126
|
+
element to be actionable, which naturally waits out overlays and transitions.
|
|
127
|
+
Reserve `force` for the rare element a component library wrongly reports as
|
|
128
|
+
disabled, and even then verify the outcome afterward.
|
|
129
|
+
|
|
130
|
+
## Error handling
|
|
131
|
+
|
|
132
|
+
Throw on unexpected failures; the harness catches and reports. Never
|
|
133
|
+
catch-and-continue to paper over a problem. Auth failures come from
|
|
134
|
+
`ctx.checkAuth()`; page-interaction failures (missing element, timeout) should
|
|
135
|
+
propagate naturally. Fix root causes, not symptoms.
|
|
136
|
+
|
|
137
|
+
**Auth resolves late in SPAs.** A single-page app often loads its shell, *then*
|
|
138
|
+
decides client-side that the session is invalid and redirects to a login page a
|
|
139
|
+
beat later. So:
|
|
140
|
+
- Do NOT call `ctx.checkAuth()` immediately after `goto` — the redirect may not
|
|
141
|
+
have happened yet (false pass) and the URL may not have settled.
|
|
142
|
+
- Do NOT race the success signal against the login URL — a valid session can
|
|
143
|
+
*transiently* touch a login-ish URL before bouncing back (false fail).
|
|
144
|
+
- DO wait for your success signal (the authenticated view's element). Only if that
|
|
145
|
+
times out, *then* call `ctx.checkAuth()` — by then the URL has settled, so a
|
|
146
|
+
login page is a real `AuthError` and anything else is a genuine render timeout.
|
|
147
|
+
|
|
148
|
+
## Naming
|
|
149
|
+
|
|
150
|
+
- Files: `verb-noun-qualifier.mjs` (e.g. `get-repo-ci-error.mjs`).
|
|
151
|
+
- Step functions: `verbNoun` camelCase (`navigateToRepo`, `dismissModals`).
|
|
152
|
+
- Log sections: short, lowercase, no spaces (`navigate`, `dismiss`, `extract`).
|
|
153
|
+
|
|
154
|
+
## Waits
|
|
155
|
+
|
|
156
|
+
Wait for **specific, verifiable things** — never for time. This is the rule that
|
|
157
|
+
makes scripts bulletproof.
|
|
158
|
+
|
|
159
|
+
### `waitForTimeout` is effectively banned
|
|
160
|
+
|
|
161
|
+
A fixed sleep is a bet that something will be ready by then. The bet loses
|
|
162
|
+
intermittently — that is precisely what flakiness *is*. Exhaust every avenue for a
|
|
163
|
+
condition-based wait before even considering a sleep:
|
|
164
|
+
|
|
165
|
+
1. Wait for an element/state that proves readiness (`locator.waitFor`,
|
|
166
|
+
`page.waitForURL`, `expect(locator).toBeVisible()`).
|
|
167
|
+
2. Wait for a content predicate via `page.waitForFunction(() => …)` when readiness
|
|
168
|
+
is "the data populated", not just "an element exists".
|
|
169
|
+
3. Wait for a network response (`page.waitForResponse`) when the DOM gives no
|
|
170
|
+
signal but a known request does.
|
|
171
|
+
4. Install a handler for interrupting UI (`page.addLocatorHandler`, below) instead
|
|
172
|
+
of sleeping to "let a modal pass".
|
|
173
|
+
|
|
174
|
+
Only if ALL of these are genuinely impossible may you fall back to
|
|
175
|
+
`page.waitForTimeout` — and then you must (a) keep it short, (b) write a comment
|
|
176
|
+
explaining what DOM-observable signal you searched for and why none exists, and
|
|
177
|
+
(c) feel bad about it. Treat each one as a defect to be removed later. A script
|
|
178
|
+
should aim for **zero** `waitForTimeout` calls.
|
|
179
|
+
|
|
180
|
+
### `networkidle` is NOT a readiness signal
|
|
181
|
+
|
|
182
|
+
`waitForLoadState('networkidle')` means "the network went quiet", which in a
|
|
183
|
+
modern SPA happens long before — or long after — the content you want renders.
|
|
184
|
+
A large SPA commonly hits network idle while the DOM is still an empty ~600-char
|
|
185
|
+
shell, with every real value still to be fetched and rendered client-side. Never
|
|
186
|
+
treat `networkidle` as "the page is ready". Wait for the *specific element or
|
|
187
|
+
text* you need instead. Use `domcontentloaded` for the initial `goto`, then a
|
|
188
|
+
concrete element wait.
|
|
189
|
+
|
|
190
|
+
### Pick a signal that proves the exact thing you need
|
|
191
|
+
|
|
192
|
+
- "Tab bar loaded" → wait for a specific named tab to be visible.
|
|
193
|
+
- "List rendered" → wait for a row's distinguishing text (e.g. a commit hash
|
|
194
|
+
pattern), not a generic container that exists while empty.
|
|
195
|
+
- "Log loaded" → wait for the log element AND a length/content predicate, so an
|
|
196
|
+
empty placeholder doesn't satisfy the wait.
|
|
197
|
+
|
|
198
|
+
### Virtualized lists/grids → set a tall viewport, don't scroll-accumulate
|
|
199
|
+
|
|
200
|
+
A virtualized list or grid renders only the rows within the scroll viewport (a
|
|
201
|
+
31-row table may put only ~17 rows in the DOM). A single DOM sweep then
|
|
202
|
+
silently returns a partial set. The cheap, robust fix is to enlarge the viewport
|
|
203
|
+
BEFORE navigating, so the grid materializes every row at once:
|
|
204
|
+
|
|
205
|
+
```js
|
|
206
|
+
await page.setViewportSize({ width: 1600, height: 20000 }); // then goto()
|
|
207
|
+
```
|
|
208
|
+
|
|
209
|
+
This overrides the harness's default 1920×1080 per-page and needs no harness change.
|
|
210
|
+
Prefer it over a scroll-accumulate loop: far less code, no timing loop. Then **verify
|
|
211
|
+
completeness** — extract the count the UI advertises (e.g. a "Properties 31" header
|
|
212
|
+
badge) and assert the extracted row count equals it, so a clipped read fails loud
|
|
213
|
+
instead of returning a silent subset. A tall viewport is not universal: a virtualizer
|
|
214
|
+
bounded by its own container's fixed CSS height can still clip regardless of window
|
|
215
|
+
size — the assertion is what catches that, and scroll-accumulate is the fallback.
|
|
216
|
+
|
|
217
|
+
### Unpredictable interrupting UI → `addLocatorHandler`, not sleeps
|
|
218
|
+
|
|
219
|
+
Modals/banners that appear at an unpredictable moment (welcome dialogs, "what's
|
|
220
|
+
new", cookie prompts) are a classic flake source: dismiss-then-continue races the
|
|
221
|
+
modal's appearance. Register a handler once; Playwright auto-runs it whenever that
|
|
222
|
+
element would block an action — fully timing-independent:
|
|
223
|
+
|
|
224
|
+
```js
|
|
225
|
+
await page.addLocatorHandler(
|
|
226
|
+
page.getByRole('dialog').filter({ has: page.getByRole('button', { name: 'Close' }) }),
|
|
227
|
+
async (dialog) => { await dialog.getByRole('button', { name: 'Close' }).click(); }
|
|
228
|
+
);
|
|
229
|
+
```
|
|
230
|
+
|
|
231
|
+
### Timeouts
|
|
232
|
+
|
|
233
|
+
Pass `ctx.timeout` to waits rather than hardcoding numbers, so a slow environment
|
|
234
|
+
can be accommodated centrally. A generous timeout on a *correct* condition is
|
|
235
|
+
fine — it only ever waits as long as it must, then proceeds the instant the
|
|
236
|
+
condition holds. That is the opposite of a fixed sleep.
|
|
237
|
+
|
|
238
|
+
### Prefer robust locators
|
|
239
|
+
|
|
240
|
+
Favor role/text/label locators (`getByRole`, `getByText`, `getByLabel`) and stable
|
|
241
|
+
attributes (`data-testid`) over brittle CSS/class chains — component-library class
|
|
242
|
+
names (`bp6-…`) change between versions. When the only distinguishing feature is
|
|
243
|
+
visible text, a text/regex locator is more durable than a guessed class.
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Chrome State Extraction
|
|
3
|
+
*
|
|
4
|
+
* Reads cookies from Chrome's SQLite DB, decrypts them using the OS keyring,
|
|
5
|
+
* and produces Playwright-compatible storageState JSON.
|
|
6
|
+
*
|
|
7
|
+
* v10 format: 'v10'(3) + ciphertext. AES-128-CBC, IV = 16 spaces.
|
|
8
|
+
* v11 format: 'v11'(3) + IV(16) + ciphertext. AES-128-CBC. Plaintext has 16-byte random prefix.
|
|
9
|
+
* Key: PBKDF2(keyring_password, 'saltysalt', 1 iteration, 16 bytes, SHA1)
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import { existsSync, readdirSync, mkdirSync, writeFileSync } from 'fs';
|
|
13
|
+
import { join } from 'path';
|
|
14
|
+
import { homedir } from 'os';
|
|
15
|
+
import { pbkdf2Sync, createDecipheriv } from 'crypto';
|
|
16
|
+
import Database from 'better-sqlite3';
|
|
17
|
+
import { getChromeSafeStoragePassword } from './keyring.mjs';
|
|
18
|
+
|
|
19
|
+
const DEFAULT_CHROME_BASE = join(homedir(), '.config', 'google-chrome');
|
|
20
|
+
const STATE_CACHE_DIR = join(homedir(), '.cache', 'browser-automation-state');
|
|
21
|
+
|
|
22
|
+
export function getChromeProfilePath(profileName = 'Default') {
|
|
23
|
+
const profileDir = join(DEFAULT_CHROME_BASE, profileName);
|
|
24
|
+
if (!existsSync(profileDir)) {
|
|
25
|
+
throw new Error(
|
|
26
|
+
`Chrome profile "${profileName}" not found at ${profileDir}. ` +
|
|
27
|
+
`Available: ${listProfiles().join(', ')}`
|
|
28
|
+
);
|
|
29
|
+
}
|
|
30
|
+
return profileDir;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export function listProfiles() {
|
|
34
|
+
if (!existsSync(DEFAULT_CHROME_BASE)) return [];
|
|
35
|
+
return readdirSync(DEFAULT_CHROME_BASE, { withFileTypes: true })
|
|
36
|
+
.filter(d => d.isDirectory() && (d.name === 'Default' || d.name.startsWith('Profile ')))
|
|
37
|
+
.map(d => d.name);
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
function deriveKey(password) {
|
|
41
|
+
return pbkdf2Sync(password, 'saltysalt', 1, 16, 'sha1');
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
function decryptCookieValue(encryptedValue, key) {
|
|
45
|
+
if (!encryptedValue || encryptedValue.length === 0) return '';
|
|
46
|
+
|
|
47
|
+
const prefix = encryptedValue.slice(0, 3).toString('utf-8');
|
|
48
|
+
|
|
49
|
+
if (prefix === 'v10') {
|
|
50
|
+
const iv = Buffer.alloc(16, ' ');
|
|
51
|
+
const ciphertext = encryptedValue.slice(3);
|
|
52
|
+
const decipher = createDecipheriv('aes-128-cbc', key, iv);
|
|
53
|
+
const dec = Buffer.concat([decipher.update(ciphertext), decipher.final()]);
|
|
54
|
+
return dec.toString('utf-8');
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
if (prefix === 'v11') {
|
|
58
|
+
const iv = encryptedValue.slice(3, 19);
|
|
59
|
+
const ciphertext = encryptedValue.slice(19);
|
|
60
|
+
const decipher = createDecipheriv('aes-128-cbc', key, iv);
|
|
61
|
+
const dec = Buffer.concat([decipher.update(ciphertext), decipher.final()]);
|
|
62
|
+
// v11 prepends 16 random bytes to the plaintext before encrypting
|
|
63
|
+
return dec.slice(16).toString('utf-8');
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
// Unencrypted or unknown format
|
|
67
|
+
return encryptedValue.toString('utf-8');
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
function chromeTimeToUnix(chromeTime) {
|
|
71
|
+
if (!chromeTime || chromeTime === 0) return -1;
|
|
72
|
+
const epochOffset = 11644473600000000n;
|
|
73
|
+
const unixMicro = BigInt(chromeTime) - epochOffset;
|
|
74
|
+
return Number(unixMicro / 1000000n);
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Read and decrypt cookies from a Chrome profile.
|
|
79
|
+
* @param {string} profileName - Chrome profile name (default: 'Default')
|
|
80
|
+
* @param {string[]|null} domains - Filter by domain substrings. Null = all cookies.
|
|
81
|
+
*/
|
|
82
|
+
export async function extractCookies(profileName = 'Default', domains = null) {
|
|
83
|
+
const profileDir = getChromeProfilePath(profileName);
|
|
84
|
+
const cookieDbPath = join(profileDir, 'Cookies');
|
|
85
|
+
|
|
86
|
+
if (!existsSync(cookieDbPath)) {
|
|
87
|
+
throw new Error(`Cookies database not found at ${cookieDbPath}`);
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
const password = await getChromeSafeStoragePassword();
|
|
91
|
+
const key = deriveKey(password);
|
|
92
|
+
|
|
93
|
+
const db = new Database(cookieDbPath, { readonly: true, fileMustExist: true });
|
|
94
|
+
|
|
95
|
+
let query = 'SELECT host_key, name, encrypted_value, path, expires_utc, is_secure, is_httponly, samesite FROM cookies';
|
|
96
|
+
const params = [];
|
|
97
|
+
|
|
98
|
+
if (domains && domains.length > 0) {
|
|
99
|
+
const clauses = domains.map(() => 'host_key LIKE ?');
|
|
100
|
+
query += ` WHERE ${clauses.join(' OR ')}`;
|
|
101
|
+
for (const d of domains) {
|
|
102
|
+
params.push(`%${d}%`);
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
const rows = db.prepare(query).all(...params);
|
|
107
|
+
db.close();
|
|
108
|
+
|
|
109
|
+
const cookies = [];
|
|
110
|
+
for (const row of rows) {
|
|
111
|
+
try {
|
|
112
|
+
const value = decryptCookieValue(row.encrypted_value, key);
|
|
113
|
+
cookies.push({
|
|
114
|
+
name: row.name,
|
|
115
|
+
value,
|
|
116
|
+
domain: row.host_key,
|
|
117
|
+
path: row.path,
|
|
118
|
+
expires: chromeTimeToUnix(row.expires_utc),
|
|
119
|
+
httpOnly: Boolean(row.is_httponly),
|
|
120
|
+
secure: Boolean(row.is_secure),
|
|
121
|
+
sameSite: ['None', 'Lax', 'Strict'][row.samesite] || 'None',
|
|
122
|
+
});
|
|
123
|
+
} catch (e) {
|
|
124
|
+
// Skip cookies that fail to decrypt (shouldn't happen but don't break the whole run)
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
return cookies;
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* Build a Playwright-compatible storageState object.
|
|
133
|
+
*/
|
|
134
|
+
export async function buildStorageState(profileName = 'Default', domains = null) {
|
|
135
|
+
const cookies = await extractCookies(profileName, domains);
|
|
136
|
+
return { cookies, origins: [] };
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* Write storageState to a JSON file and return the path.
|
|
141
|
+
*/
|
|
142
|
+
export async function saveStorageState(profileName = 'Default', domains = null) {
|
|
143
|
+
mkdirSync(STATE_CACHE_DIR, { recursive: true });
|
|
144
|
+
const state = await buildStorageState(profileName, domains);
|
|
145
|
+
const outPath = join(STATE_CACHE_DIR, `storage-state-${profileName.replace(/\s+/g, '-')}.json`);
|
|
146
|
+
writeFileSync(outPath, JSON.stringify(state, null, 2));
|
|
147
|
+
return outPath;
|
|
148
|
+
}
|