@sous-io/sous 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/LICENSE +201 -0
  2. package/README.md +154 -0
  3. package/bin/run.js +17 -0
  4. package/bin/xcv +5 -0
  5. package/package.json +81 -0
  6. package/shared-prompts/_partials/resume-task.md +51 -0
  7. package/shared-prompts/_partials/sub-agent-delegation.md +32 -0
  8. package/shared-prompts/_partials/update-task-file.md +52 -0
  9. package/shared-prompts/memories/automated-browser-tasks/INDEX.tpl.md +52 -0
  10. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/SKILL.tpl.md +102 -0
  11. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/examples/auth-failure-handling.mjs +81 -0
  12. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/examples/chained-workflow.mjs +126 -0
  13. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/examples/simple-fetch.mjs +92 -0
  14. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/references/architecture.md +61 -0
  15. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/references/auth-and-sessions.md +65 -0
  16. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/references/ctx-api.md +96 -0
  17. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/references/installation.md +104 -0
  18. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/references/script-conventions.md +243 -0
  19. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/chrome-state.mjs +148 -0
  20. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/debug.mjs +383 -0
  21. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/debug.spec.mjs +267 -0
  22. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/eslint.config.mjs +56 -0
  23. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/harness.mjs +169 -0
  24. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/keyring.mjs +59 -0
  25. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/logger.mjs +25 -0
  26. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/params.mjs +140 -0
  27. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/run.mjs +140 -0
  28. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/settings.tpl.mjs +1 -0
  29. package/shared-prompts/skills/automated-browser-tasks/about-automated-browser-tasks/scripts/utils.mjs +185 -0
  30. package/shared-prompts/skills/automated-browser-tasks/create-automated-browser-task/SKILL.tpl.md +52 -0
  31. package/shared-prompts/skills/automated-browser-tasks/running-automated-browser-tasks/SKILL.tpl.md +59 -0
  32. package/shared-prompts/skills/automated-browser-tasks/update-automated-browser-task/SKILL.tpl.md +47 -0
  33. package/shared-prompts/skills/control-flow/approve/SKILL.tpl.md +26 -0
  34. package/shared-prompts/skills/control-flow/opine/SKILL.tpl.md +58 -0
  35. package/shared-prompts/skills/control-flow/repeat/SKILL.tpl.md +27 -0
  36. package/shared-prompts/skills/control-flow/research/SKILL.tpl.md +34 -0
  37. package/shared-prompts/skills/sous-skills/about-agent-skills/SKILL.tpl.md +177 -0
  38. package/shared-prompts/skills/sous-skills/about-agent-skills/examples/about-something.md +45 -0
  39. package/shared-prompts/skills/sous-skills/about-agent-skills/examples/do-something.md +33 -0
  40. package/shared-prompts/skills/sous-skills/about-agent-skills/references/advanced-patterns.md +87 -0
  41. package/shared-prompts/skills/sous-skills/about-agent-skills/references/commands.md +46 -0
  42. package/shared-prompts/skills/sous-skills/about-agent-skills/references/frontmatter.md +25 -0
  43. package/shared-prompts/skills/sous-skills/about-agent-skills/references/substitutions.md +50 -0
  44. package/shared-prompts/skills/sous-skills/about-liquid-templates/SKILL.tpl.md +268 -0
  45. package/shared-prompts/skills/sous-skills/about-liquid-templates/references/liquid-filters.md +82 -0
  46. package/shared-prompts/skills/sous-skills/about-sous/SKILL.tpl.md +51 -0
  47. package/shared-prompts/skills/sous-skills/create-skill/SKILL.tpl.md +114 -0
  48. package/shared-prompts/skills/task-files/about-task-files/SKILL.tpl.md +122 -0
  49. package/shared-prompts/skills/task-files/continue-task-in-new-branch/SKILL.tpl.md +80 -0
  50. package/shared-prompts/skills/task-files/go/SKILL.tpl.md +14 -0
  51. package/shared-prompts/skills/task-files/resume-task/SKILL.tpl.md +13 -0
  52. package/shared-prompts/skills/task-files/start-task/SKILL.tpl.md +93 -0
  53. package/shared-prompts/skills/task-files/update/SKILL.tpl.md +14 -0
  54. package/shared-prompts/skills/task-files/update-task-file/SKILL.tpl.md +13 -0
  55. package/src/base-command.ts +163 -0
  56. package/src/commands/build.ts +196 -0
  57. package/src/commands/clear.ts +71 -0
  58. package/src/commands/compile.ts +95 -0
  59. package/src/commands/launch.ts +111 -0
  60. package/src/commands/prune.ts +48 -0
  61. package/src/lib/build-service.ts +258 -0
  62. package/src/lib/config-discovery.ts +199 -0
  63. package/src/lib/env-local.ts +195 -0
  64. package/src/lib/include-resolver.ts +146 -0
  65. package/src/lib/markdown-compiler.ts +580 -0
  66. package/src/lib/pid-service.ts +88 -0
  67. package/src/lib/settings.ts +695 -0
  68. package/src/lib/state.ts +135 -0
  69. package/src/lib/watch-service.ts +115 -0
  70. package/src/templating/filters/bullet-list.ts +9 -0
  71. package/src/templating/filters/index.ts +8 -0
  72. package/src/templating/init-liquid-engine.ts +82 -0
  73. package/src/templating/lib/glob-files.ts +74 -0
  74. package/src/templating/lib/import-export.ts +32 -0
  75. package/src/templating/lib/tag-args.ts +19 -0
  76. package/src/templating/tags/exportScalarVarsJs.ts +43 -0
  77. package/src/templating/tags/getFiles.ts +89 -0
  78. package/src/templating/tags/index.ts +14 -0
  79. package/src/templating/tags/listFiles.ts +54 -0
  80. package/src/templating/tags/showVars.ts +22 -0
  81. package/src/utils/formatting.ts +338 -0
  82. package/src/utils/prompts.ts +19 -0
@@ -0,0 +1,104 @@
1
+ # Installation
2
+
3
+ ## Runtime & platform
4
+
5
+ - Node.js ≥ 22.
6
+ - Linux with a GNOME-keyring-compatible Secret Service (the user's Chrome must
7
+ have stored its Safe Storage key there — true after Chrome has run once on a
8
+ desktop session with an unlocked keyring).
9
+ - Google Chrome installed with at least one profile the user has logged into.
10
+
11
+ macOS (Keychain) and Windows (DPAPI) are not yet supported by `keyring.mjs`.
12
+
13
+ ## Dependencies
14
+
15
+ The runner and harness import these at runtime; the framework does not bundle
16
+ them. Install them **at the consuming project's root** — NOT globally.
17
+
18
+ Why not global: the scripts use ESM `import 'playwright'`. ESM resolves bare
19
+ imports by walking *up* the directory tree from the importing file looking for a
20
+ `node_modules`. The compiled runner lives at
21
+ `<projectRoot>/.claude/skills/about-automated-browser-tasks/scripts/run.mjs`, so a
22
+ `node_modules` at `<projectRoot>` is found by walking up; a global npm install is
23
+ never on that resolution path.
24
+
25
+ ```bash
26
+ cd <projectRoot> # the repo root that contains .claude/skills/
27
+ npm install playwright better-sqlite3 dbus-next
28
+ npx playwright install chromium
29
+ ```
30
+
31
+ Add a `package.json` at `<projectRoot>` if none exists (`{"type":"module","private":true}`)
32
+ and gitignore `node_modules/`.
33
+
34
+ - `playwright` — headless browser automation.
35
+ - `better-sqlite3` — reads Chrome's `Cookies` SQLite DB.
36
+ - `dbus-next` — pure-JS D-Bus client for the keyring (no Python, no native build).
37
+
38
+ Tested with: Playwright 1.61, better-sqlite3 12.x, Node 22, Chrome cookie format
39
+ v11, Ubuntu 22.04.
40
+
41
+ ## Project wiring (via sous)
42
+
43
+ A downstream project compiles this bundle into its skills directory and compiles
44
+ `settings.tpl.mjs` → `settings.mjs` (sibling of `run.mjs`) so scripts get
45
+ `ctx.settings`. Example compilation targets:
46
+
47
+ ```js
48
+ compilation: {
49
+ targets: [
50
+ {
51
+ // The skills (SKILL.tpl.md, references, examples, scripts) → skills dir
52
+ entryGlob: "${sousRootPath}/shared-prompts/skills/automated-browser-tasks/**/*",
53
+ outputs: [{ destinationDir: "${projectRoot}/.claude/skills" }],
54
+ },
55
+ ],
56
+ }
57
+ ```
58
+
59
+ Define project values (`chromeProfile`, base URLs, resource IDs, …) in `_vars`. The
60
+ `{% exportScalarVarsJs %}` tag in `settings.tpl.mjs` emits all in-scope scalars
61
+ as the runtime settings module — no per-key wiring needed. Be sure to define
62
+ `browserAutomationScriptsDir` (the absolute path to the project's task scripts)
63
+ in `_vars` — both the runtime and the task manifest below rely on it.
64
+
65
+ ## Task manifest in core memory
66
+
67
+ So the agent always knows which browser tasks exist (without relying on a skill
68
+ trigger firing), render the shared memory partial into the project's memory source
69
+ tree, then `@include` it from a core-memory file. It renders a live list of every
70
+ task script via `{% getFiles … import="meta" %}`, reading each script's `meta`.
71
+
72
+ `@include` does NOT substitute variables, so you cannot `@`-include the shared
73
+ `INDEX.tpl.md` by an absolute `${...}` path. Instead, add a compilation target that
74
+ renders it into your memory tree (exactly how `runtimeContext` emits
75
+ `session-context.md`):
76
+
77
+ ```js
78
+ // A target that renders the shared manifest into the project's memory source.
79
+ const browserTaskManifest = {
80
+ entryPoint: "${sousRootPath}/shared-prompts/memories/automated-browser-tasks/INDEX.tpl.md",
81
+ outputs: [
82
+ { destinationFile: "${memoryRoot}/tools/automated-browser-tasks.md" },
83
+ ],
84
+ };
85
+ ```
86
+
87
+ Order this target BEFORE the memories target that composes core memory. Then pull
88
+ it into a memory file (e.g. `tools/README.md`) with a plain relative Sous include:
89
+ put an `@`-prefixed line containing just the rendered filename
90
+ (`automated-browser-tasks.md`) on its own line in that file.
91
+
92
+ The manifest auto-rebuilds on every `xcv build`, so newly created tasks appear
93
+ automatically. It requires `browserAutomationScriptsDir` to be in scope (the
94
+ absolute path to the task scripts).
95
+
96
+ ## Verifying
97
+
98
+ Run any example script by absolute path:
99
+
100
+ ```bash
101
+ node <scriptsDir>/run.mjs <scriptsDir>/../examples/simple-fetch.mjs --url=https://example.com
102
+ ```
103
+
104
+ A clean run prints extracted cookie counts, a browser-ready line, and the result.
@@ -0,0 +1,243 @@
1
+ # Script Conventions
2
+
3
+ These rules are non-negotiable. A reviewer (or linter) should be able to reject a
4
+ script that violates them.
5
+
6
+ ## Quality bar: bulletproof or it doesn't ship
7
+
8
+ A flaky script is a broken script. "Works most of the time" is failure. Write for
9
+ 100% reliability across many consecutive and parallel runs from the first draft —
10
+ do not ship something that "usually works" and plan to harden later.
11
+
12
+ The single greatest source of flakiness is **guessing about timing instead of
13
+ waiting for facts**. Every wait must key off a concrete, observable condition that
14
+ *proves* the thing you need is ready. Spend the extra time to find that signal.
15
+ Arbitrary delays (`waitForTimeout`) are the enemy — see [Waits](#waits); they are
16
+ effectively banned.
17
+
18
+ Before writing a single wait or selector, **observe the real page.** Do not assume
19
+ DOM structure. Build your waits from what you actually see — selectors invented
20
+ from imagination are how you get a script that passes once and fails in CI.
21
+
22
+ `ctx.debug` exists for exactly this (full surface in `ctx-api.md`):
23
+
24
+ - `ctx.debug.dump()` — snapshot URL, title, text, screenshot, and HTML to disk.
25
+ - `ctx.debug.describe(selector)` / `ctx.debug.clickables()` — see whether a
26
+ selector matches and what the real interactive elements are (often a
27
+ click-handled `div`, not the `<a>`/`<button>` you assumed).
28
+ - `ctx.debug.findText(text)` — locate text and the clickable ancestor to target.
29
+ - `ctx.debug.watch(() => metric)` — sample a metric over time to find the *moment*
30
+ the page is genuinely ready (this is how you discover that `networkidle` fired
31
+ on an empty shell), then wait on that concrete signal.
32
+
33
+ On an unexpected throw, the harness auto-captures a failure snapshot
34
+ (`result.debugDir`) — check it first when a run fails. Remove file-writing debug
35
+ calls (`dump`/`screenshot`/`html`) once the script is solid; keep them out of hot
36
+ loops.
37
+
38
+ ## Structure
39
+
40
+ `execute(ctx)` is a thin orchestrator that reads like a table of contents. All
41
+ real work lives in small, named step functions defined below `execute` in the
42
+ same file.
43
+
44
+ - One discrete action per step function (navigate, dismiss, extract, parse…).
45
+ - ≤ 30 lines per function; 10 or fewer is ideal.
46
+ - Module-level functions, not class methods.
47
+ - Generic patterns → `ctx.utils`. Site-specific patterns → step functions (which
48
+ a project may later factor into shared libs it imports).
49
+
50
+ ## Doc-blocks
51
+
52
+ EVERY function — `execute` included — has a proper JSDoc block: a description
53
+ line plus `@param` for every argument and `@returns`. Use
54
+ `@returns {Promise<void>}` for functions that return nothing. Single-line
55
+ `/** … */` comments are NOT sufficient.
56
+
57
+ ## Destructuring
58
+
59
+ Each function destructures the members it needs off `ctx` (and off `params`) at
60
+ the top of its body, so the body never repeats `ctx.`/`params.` prefixes:
61
+
62
+ ```js
63
+ async function navigateToRepo(ctx, baseUrl, repoId) {
64
+ const { page, logger, timeout, checkAuth } = ctx;
65
+ ...
66
+ }
67
+ ```
68
+
69
+ ## Params (`meta.params`)
70
+
71
+ The framework resolves and validates params before `execute` runs. Scripts never
72
+ validate their own params. Resolution priority (low → high):
73
+ `ctx.settings` < `meta.params[x].default` < explicit (CLI) params.
74
+
75
+ Each param spec:
76
+
77
+ | Field | Type | Meaning |
78
+ |-------|------|---------|
79
+ | `required` | boolean | Error if nothing resolves. |
80
+ | `default` | any | Fallback value. |
81
+ | `description` | string | Be genuinely descriptive: what it is, where to find it, how it's used, consequence of omitting. Shown in listings and errors. |
82
+ | `validate` | `RegExp` \| `Function` | See below. |
83
+ | `invalidMessage` | string | Error for a failing RegExp, or a `validate` fn returning `false`. |
84
+
85
+ `validate`:
86
+ - **RegExp** — resolved value (as string) must match.
87
+ - **Function** `(value, resolvedParams) => true | false | string` — `true` valid;
88
+ a returned `string` is used as the error; `false` falls back to `invalidMessage`.
89
+ The function gets all resolved params, enabling cross-param checks.
90
+
91
+ All failures across params are collected into one `ParamError`.
92
+
93
+ ## Logging
94
+
95
+ Use `ctx.logger`, never `console.log`. Create a child per section:
96
+ `const log = ctx.logger.child('navigate')`. Output is
97
+ `[script-name:section] message`. Levels: `info`, `warn`, `error`.
98
+
99
+ ## Return shape
100
+
101
+ Return a plain object. Common keys:
102
+
103
+ - `found: boolean` — whether the target content was located.
104
+ - `content: string` — extracted content (when found).
105
+ - `outputFile: string` — path for the runner to write `content` to.
106
+ - a URL key (e.g. `buildUrl`) — where content was found.
107
+ - `message: string` — human-readable explanation, especially on failure.
108
+
109
+ On `found: false`, include diagnostics (`pageTextPreview`, `message`).
110
+
111
+ ## Verify every action
112
+
113
+ Do not assume an action took effect — prove it. After every navigation or click
114
+ that changes state, wait for a signal that confirms the *intended outcome*:
115
+
116
+ - After a navigation: `await page.waitForURL(/expected-path/)`, or wait for an
117
+ element that only exists on the destination.
118
+ - After a click that should open a view: wait for that view's content, not just
119
+ for the click to return.
120
+ - After triggering content load: wait for the content to be present AND non-empty
121
+ (e.g. a `<pre>` whose text length exceeds a threshold), not merely attached.
122
+
123
+ A click with `{ force: true }` is fire-and-forget: it bypasses Playwright's
124
+ actionability checks (visible, stable, not covered) and reports success even when
125
+ it lands on nothing. Prefer a plain click — Playwright then auto-waits for the
126
+ element to be actionable, which naturally waits out overlays and transitions.
127
+ Reserve `force` for the rare element a component library wrongly reports as
128
+ disabled, and even then verify the outcome afterward.
129
+
130
+ ## Error handling
131
+
132
+ Throw on unexpected failures; the harness catches and reports. Never
133
+ catch-and-continue to paper over a problem. Auth failures come from
134
+ `ctx.checkAuth()`; page-interaction failures (missing element, timeout) should
135
+ propagate naturally. Fix root causes, not symptoms.
136
+
137
+ **Auth resolves late in SPAs.** A single-page app often loads its shell, *then*
138
+ decides client-side that the session is invalid and redirects to a login page a
139
+ beat later. So:
140
+ - Do NOT call `ctx.checkAuth()` immediately after `goto` — the redirect may not
141
+ have happened yet (false pass) and the URL may not have settled.
142
+ - Do NOT race the success signal against the login URL — a valid session can
143
+ *transiently* touch a login-ish URL before bouncing back (false fail).
144
+ - DO wait for your success signal (the authenticated view's element). Only if that
145
+ times out, *then* call `ctx.checkAuth()` — by then the URL has settled, so a
146
+ login page is a real `AuthError` and anything else is a genuine render timeout.
147
+
148
+ ## Naming
149
+
150
+ - Files: `verb-noun-qualifier.mjs` (e.g. `get-repo-ci-error.mjs`).
151
+ - Step functions: `verbNoun` camelCase (`navigateToRepo`, `dismissModals`).
152
+ - Log sections: short, lowercase, no spaces (`navigate`, `dismiss`, `extract`).
153
+
154
+ ## Waits
155
+
156
+ Wait for **specific, verifiable things** — never for time. This is the rule that
157
+ makes scripts bulletproof.
158
+
159
+ ### `waitForTimeout` is effectively banned
160
+
161
+ A fixed sleep is a bet that something will be ready by then. The bet loses
162
+ intermittently — that is precisely what flakiness *is*. Exhaust every avenue for a
163
+ condition-based wait before even considering a sleep:
164
+
165
+ 1. Wait for an element/state that proves readiness (`locator.waitFor`,
166
+ `page.waitForURL`, `expect(locator).toBeVisible()`).
167
+ 2. Wait for a content predicate via `page.waitForFunction(() => …)` when readiness
168
+ is "the data populated", not just "an element exists".
169
+ 3. Wait for a network response (`page.waitForResponse`) when the DOM gives no
170
+ signal but a known request does.
171
+ 4. Install a handler for interrupting UI (`page.addLocatorHandler`, below) instead
172
+ of sleeping to "let a modal pass".
173
+
174
+ Only if ALL of these are genuinely impossible may you fall back to
175
+ `page.waitForTimeout` — and then you must (a) keep it short, (b) write a comment
176
+ explaining what DOM-observable signal you searched for and why none exists, and
177
+ (c) feel bad about it. Treat each one as a defect to be removed later. A script
178
+ should aim for **zero** `waitForTimeout` calls.
179
+
180
+ ### `networkidle` is NOT a readiness signal
181
+
182
+ `waitForLoadState('networkidle')` means "the network went quiet", which in a
183
+ modern SPA happens long before — or long after — the content you want renders.
184
+ A large SPA commonly hits network idle while the DOM is still an empty ~600-char
185
+ shell, with every real value still to be fetched and rendered client-side. Never
186
+ treat `networkidle` as "the page is ready". Wait for the *specific element or
187
+ text* you need instead. Use `domcontentloaded` for the initial `goto`, then a
188
+ concrete element wait.
189
+
190
+ ### Pick a signal that proves the exact thing you need
191
+
192
+ - "Tab bar loaded" → wait for a specific named tab to be visible.
193
+ - "List rendered" → wait for a row's distinguishing text (e.g. a commit hash
194
+ pattern), not a generic container that exists while empty.
195
+ - "Log loaded" → wait for the log element AND a length/content predicate, so an
196
+ empty placeholder doesn't satisfy the wait.
197
+
198
+ ### Virtualized lists/grids → set a tall viewport, don't scroll-accumulate
199
+
200
+ A virtualized list or grid renders only the rows within the scroll viewport (a
201
+ 31-row table may put only ~17 rows in the DOM). A single DOM sweep then
202
+ silently returns a partial set. The cheap, robust fix is to enlarge the viewport
203
+ BEFORE navigating, so the grid materializes every row at once:
204
+
205
+ ```js
206
+ await page.setViewportSize({ width: 1600, height: 20000 }); // then goto()
207
+ ```
208
+
209
+ This overrides the harness's default 1920×1080 per-page and needs no harness change.
210
+ Prefer it over a scroll-accumulate loop: far less code, no timing loop. Then **verify
211
+ completeness** — extract the count the UI advertises (e.g. a "Properties 31" header
212
+ badge) and assert the extracted row count equals it, so a clipped read fails loud
213
+ instead of returning a silent subset. A tall viewport is not universal: a virtualizer
214
+ bounded by its own container's fixed CSS height can still clip regardless of window
215
+ size — the assertion is what catches that, and scroll-accumulate is the fallback.
216
+
217
+ ### Unpredictable interrupting UI → `addLocatorHandler`, not sleeps
218
+
219
+ Modals/banners that appear at an unpredictable moment (welcome dialogs, "what's
220
+ new", cookie prompts) are a classic flake source: dismiss-then-continue races the
221
+ modal's appearance. Register a handler once; Playwright auto-runs it whenever that
222
+ element would block an action — fully timing-independent:
223
+
224
+ ```js
225
+ await page.addLocatorHandler(
226
+ page.getByRole('dialog').filter({ has: page.getByRole('button', { name: 'Close' }) }),
227
+ async (dialog) => { await dialog.getByRole('button', { name: 'Close' }).click(); }
228
+ );
229
+ ```
230
+
231
+ ### Timeouts
232
+
233
+ Pass `ctx.timeout` to waits rather than hardcoding numbers, so a slow environment
234
+ can be accommodated centrally. A generous timeout on a *correct* condition is
235
+ fine — it only ever waits as long as it must, then proceeds the instant the
236
+ condition holds. That is the opposite of a fixed sleep.
237
+
238
+ ### Prefer robust locators
239
+
240
+ Favor role/text/label locators (`getByRole`, `getByText`, `getByLabel`) and stable
241
+ attributes (`data-testid`) over brittle CSS/class chains — component-library class
242
+ names (`bp6-…`) change between versions. When the only distinguishing feature is
243
+ visible text, a text/regex locator is more durable than a guessed class.
@@ -0,0 +1,148 @@
1
+ /**
2
+ * Chrome State Extraction
3
+ *
4
+ * Reads cookies from Chrome's SQLite DB, decrypts them using the OS keyring,
5
+ * and produces Playwright-compatible storageState JSON.
6
+ *
7
+ * v10 format: 'v10'(3) + ciphertext. AES-128-CBC, IV = 16 spaces.
8
+ * v11 format: 'v11'(3) + IV(16) + ciphertext. AES-128-CBC. Plaintext has 16-byte random prefix.
9
+ * Key: PBKDF2(keyring_password, 'saltysalt', 1 iteration, 16 bytes, SHA1)
10
+ */
11
+
12
+ import { existsSync, readdirSync, mkdirSync, writeFileSync } from 'fs';
13
+ import { join } from 'path';
14
+ import { homedir } from 'os';
15
+ import { pbkdf2Sync, createDecipheriv } from 'crypto';
16
+ import Database from 'better-sqlite3';
17
+ import { getChromeSafeStoragePassword } from './keyring.mjs';
18
+
19
+ const DEFAULT_CHROME_BASE = join(homedir(), '.config', 'google-chrome');
20
+ const STATE_CACHE_DIR = join(homedir(), '.cache', 'browser-automation-state');
21
+
22
+ export function getChromeProfilePath(profileName = 'Default') {
23
+ const profileDir = join(DEFAULT_CHROME_BASE, profileName);
24
+ if (!existsSync(profileDir)) {
25
+ throw new Error(
26
+ `Chrome profile "${profileName}" not found at ${profileDir}. ` +
27
+ `Available: ${listProfiles().join(', ')}`
28
+ );
29
+ }
30
+ return profileDir;
31
+ }
32
+
33
+ export function listProfiles() {
34
+ if (!existsSync(DEFAULT_CHROME_BASE)) return [];
35
+ return readdirSync(DEFAULT_CHROME_BASE, { withFileTypes: true })
36
+ .filter(d => d.isDirectory() && (d.name === 'Default' || d.name.startsWith('Profile ')))
37
+ .map(d => d.name);
38
+ }
39
+
40
+ function deriveKey(password) {
41
+ return pbkdf2Sync(password, 'saltysalt', 1, 16, 'sha1');
42
+ }
43
+
44
+ function decryptCookieValue(encryptedValue, key) {
45
+ if (!encryptedValue || encryptedValue.length === 0) return '';
46
+
47
+ const prefix = encryptedValue.slice(0, 3).toString('utf-8');
48
+
49
+ if (prefix === 'v10') {
50
+ const iv = Buffer.alloc(16, ' ');
51
+ const ciphertext = encryptedValue.slice(3);
52
+ const decipher = createDecipheriv('aes-128-cbc', key, iv);
53
+ const dec = Buffer.concat([decipher.update(ciphertext), decipher.final()]);
54
+ return dec.toString('utf-8');
55
+ }
56
+
57
+ if (prefix === 'v11') {
58
+ const iv = encryptedValue.slice(3, 19);
59
+ const ciphertext = encryptedValue.slice(19);
60
+ const decipher = createDecipheriv('aes-128-cbc', key, iv);
61
+ const dec = Buffer.concat([decipher.update(ciphertext), decipher.final()]);
62
+ // v11 prepends 16 random bytes to the plaintext before encrypting
63
+ return dec.slice(16).toString('utf-8');
64
+ }
65
+
66
+ // Unencrypted or unknown format
67
+ return encryptedValue.toString('utf-8');
68
+ }
69
+
70
+ function chromeTimeToUnix(chromeTime) {
71
+ if (!chromeTime || chromeTime === 0) return -1;
72
+ const epochOffset = 11644473600000000n;
73
+ const unixMicro = BigInt(chromeTime) - epochOffset;
74
+ return Number(unixMicro / 1000000n);
75
+ }
76
+
77
+ /**
78
+ * Read and decrypt cookies from a Chrome profile.
79
+ * @param {string} profileName - Chrome profile name (default: 'Default')
80
+ * @param {string[]|null} domains - Filter by domain substrings. Null = all cookies.
81
+ */
82
+ export async function extractCookies(profileName = 'Default', domains = null) {
83
+ const profileDir = getChromeProfilePath(profileName);
84
+ const cookieDbPath = join(profileDir, 'Cookies');
85
+
86
+ if (!existsSync(cookieDbPath)) {
87
+ throw new Error(`Cookies database not found at ${cookieDbPath}`);
88
+ }
89
+
90
+ const password = await getChromeSafeStoragePassword();
91
+ const key = deriveKey(password);
92
+
93
+ const db = new Database(cookieDbPath, { readonly: true, fileMustExist: true });
94
+
95
+ let query = 'SELECT host_key, name, encrypted_value, path, expires_utc, is_secure, is_httponly, samesite FROM cookies';
96
+ const params = [];
97
+
98
+ if (domains && domains.length > 0) {
99
+ const clauses = domains.map(() => 'host_key LIKE ?');
100
+ query += ` WHERE ${clauses.join(' OR ')}`;
101
+ for (const d of domains) {
102
+ params.push(`%${d}%`);
103
+ }
104
+ }
105
+
106
+ const rows = db.prepare(query).all(...params);
107
+ db.close();
108
+
109
+ const cookies = [];
110
+ for (const row of rows) {
111
+ try {
112
+ const value = decryptCookieValue(row.encrypted_value, key);
113
+ cookies.push({
114
+ name: row.name,
115
+ value,
116
+ domain: row.host_key,
117
+ path: row.path,
118
+ expires: chromeTimeToUnix(row.expires_utc),
119
+ httpOnly: Boolean(row.is_httponly),
120
+ secure: Boolean(row.is_secure),
121
+ sameSite: ['None', 'Lax', 'Strict'][row.samesite] || 'None',
122
+ });
123
+ } catch (e) {
124
+ // Skip cookies that fail to decrypt (shouldn't happen but don't break the whole run)
125
+ }
126
+ }
127
+
128
+ return cookies;
129
+ }
130
+
131
+ /**
132
+ * Build a Playwright-compatible storageState object.
133
+ */
134
+ export async function buildStorageState(profileName = 'Default', domains = null) {
135
+ const cookies = await extractCookies(profileName, domains);
136
+ return { cookies, origins: [] };
137
+ }
138
+
139
+ /**
140
+ * Write storageState to a JSON file and return the path.
141
+ */
142
+ export async function saveStorageState(profileName = 'Default', domains = null) {
143
+ mkdirSync(STATE_CACHE_DIR, { recursive: true });
144
+ const state = await buildStorageState(profileName, domains);
145
+ const outPath = join(STATE_CACHE_DIR, `storage-state-${profileName.replace(/\s+/g, '-')}.json`);
146
+ writeFileSync(outPath, JSON.stringify(state, null, 2));
147
+ return outPath;
148
+ }