@trawlme/cli 3.11.0 → 3.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -113,9 +113,12 @@ trawl scraps account delete <id> [--force] [--json]
113
113
  trawl scraps account clear-session <id> [--json]
114
114
  trawl scraps account status <id> [--json]
115
115
  trawl scraps account session set <id> -c <file> [--json]
116
+ trawl scraps account session capture <id> [--chrome <path>] [--json]
116
117
  ```
117
118
 
118
- `account session set` uploads a Puppeteer cookie JSON array to bootstrap a logged-in session without storing credentials (BYO-cookies). A missing `-u/--username`/`-p/--password` on `account set` follows the same [non-interactive rule](#non-interactive-rule) as `login`.
119
+ `account session set` uploads a cookie JSON array — or a `{ cookies, origins }` storageState file (see `session capture` below) — to bootstrap a logged-in session without storing credentials (BYO-cookies). A missing `-u/--username`/`-p/--password` on `account set` follows the same [non-interactive rule](#non-interactive-rule) as `login`.
120
+
121
+ `account session capture <id>` is the one-command version of the same idea: it opens a real, visible Chrome window at the scrap's target URL, waits for **you** to log in there — 2FA included — then reads the resulting session over the Chrome DevTools Protocol (cookies **and** per-origin localStorage) and uploads it. You stay authenticated as yourself; Trawl never sees your credentials, and responsibility for lawful use of the captured session stays with you. The capture is scoped to the scrap's own target host by RFC 6265 domain-matching, the same rule a real browser uses to decide which cookies to send: a cookie explicitly scoped to a domain (e.g. `Domain=example.com`) is in scope for that host and its subdomains, while a host-only cookie (no `Domain` attribute) is in scope only for the exact host it was set on — never a sibling subdomain, even one sharing a parent domain, and never another tenant on the same multi-tenant hosting domain (e.g. a different `*.github.io` site). localStorage follows the same anchor: an origin is in scope only if it's the target host itself or a subdomain of it. The one deliberate gap: a host-only cookie set on a sibling host (e.g. `auth.example.com` while the scrap targets `app.example.com`) is excluded — a real browser would never send it to the target host either, so nothing replay-relevant is lost unless the scrap script itself later navigates to that sibling. Completion is explicit: press Enter in the terminal once you're signed in — the primary path, and the only one guaranteed cross-platform. Closing the Chrome window also completes the capture (cookies only, not localStorage) when Chrome itself stays running after its last window closes, which this CLI verified on macOS; on a platform where closing the last window quits Chrome entirely, the command reports that fact instead of a session, and pressing Enter is the reliable path there. This command needs a local interactive terminal with a real display; it does not work headless, in CI, or over a plain SSH session, and auto-detects a system Chrome/Chromium (override with `--chrome <path>` or `TRAWL_CHROME_PATH`). Never prints the captured session itself, in any mode — `--json` echoes only the server's response plus capture counts.
119
122
 
120
123
  ### Claude Code skills
121
124
 
@@ -12,6 +12,8 @@ import { formatDoctor, formatAutofix, fetchRunAndFix, pickRun, pickFix, isAuthWa
12
12
  import { renderPinch, pinchEnabled } from '../lib/pinch.js';
13
13
  import { maybeShowReferralTip } from '../lib/tips.js';
14
14
  import { maybeSuggestSkillsForAuthWall } from '../lib/skillsNudge.js';
15
+ import { captureSession, checkInteractiveEnvironment, LOCALSTORAGE_READ_TIMEOUT_MS, } from '../lib/session-capture.js';
16
+ import { assertSecureTransport } from '../lib/secure-transport.js';
15
17
  import { getApiUrl } from '../lib/config.js';
16
18
  import { resolveDocsUrls, resolveFailureKindDocsUrl } from '../lib/docs.js';
17
19
  /**
@@ -1595,32 +1597,69 @@ account
1595
1597
  const accountSession = account
1596
1598
  .command('session')
1597
1599
  .description('Manage scrap account session cookies (flavour B BYO-cookies)');
1600
+ /** Structural check only (bare origin format, {name,value} shape) — the
1601
+ * server's sessionShape.js does the authoritative validation. No domain
1602
+ * scoping here: unlike `capture`'s CDP path, a hand-authored/exported file
1603
+ * has no "target URL" to scope against, and the human supplying it already
1604
+ * chose what to include. */
1605
+ function isValidOriginsShape(value) {
1606
+ return (Array.isArray(value) &&
1607
+ value.every((o) => o &&
1608
+ typeof o === 'object' &&
1609
+ typeof o.origin === 'string' &&
1610
+ Array.isArray(o.localStorage) &&
1611
+ o.localStorage.every((kv) => kv && typeof kv === 'object' && typeof kv.name === 'string' && typeof kv.value === 'string')));
1612
+ }
1598
1613
  // account session set
1599
1614
  accountSession
1600
1615
  .command('set <id>')
1601
- .description('Upload browser session cookies for a scrap (Puppeteer cookie JSON array)')
1602
- .requiredOption('-c, --cookies <file>', 'Path to a Puppeteer cookie JSON array file')
1616
+ .description('Upload a browser session for a scrap — a cookie JSON array, or a { cookies, origins } storageState file (see `session capture`)')
1617
+ .requiredOption('-c, --cookies <file>', 'Path to a cookie JSON array, or a storageState file ({ cookies, origins? })')
1603
1618
  .option('--json', 'Output as JSON')
1604
1619
  .action(async (id, opts) => {
1605
1620
  validateObjectId(id);
1621
+ // A session is a bearer-equivalent secret — refuse to PUT it over
1622
+ // plain HTTP (trawl_cli#183 review finding 5). Checked before anything
1623
+ // else in this action, including the file read below.
1624
+ const secureTransportError = assertSecureTransport(getApiUrl());
1625
+ if (secureTransportError) {
1626
+ usageError(secureTransportError, { json: opts.json });
1627
+ return;
1628
+ }
1606
1629
  const { existsSync, readFileSync } = await import('fs');
1607
1630
  if (!existsSync(opts.cookies)) {
1608
1631
  usageError(`File not found: ${opts.cookies}`, { json: opts.json });
1609
1632
  return;
1610
1633
  }
1611
1634
  let cookies;
1635
+ let origins;
1612
1636
  try {
1613
1637
  const raw = readFileSync(opts.cookies, 'utf-8');
1614
- cookies = JSON.parse(raw);
1638
+ const parsed = JSON.parse(raw);
1639
+ if (Array.isArray(parsed)) {
1640
+ cookies = parsed;
1641
+ }
1642
+ else if (parsed && typeof parsed === 'object' && Array.isArray(parsed.cookies)) {
1643
+ // storageState-shaped file — trawl_cli#183's `session capture` output shape.
1644
+ const shaped = parsed;
1645
+ cookies = shaped.cookies;
1646
+ if (shaped.origins !== undefined) {
1647
+ if (!isValidOriginsShape(shaped.origins)) {
1648
+ usageError('origins must be an array of { origin: string, localStorage: [{name, value}] }', { json: opts.json });
1649
+ return;
1650
+ }
1651
+ origins = shaped.origins;
1652
+ }
1653
+ }
1654
+ else {
1655
+ usageError('Cookies file must contain a JSON array, or a { cookies, origins? } storageState object', { json: opts.json });
1656
+ return;
1657
+ }
1615
1658
  }
1616
1659
  catch (e) {
1617
1660
  usageError(`Failed to parse cookies file: ${e.message}`, { json: opts.json });
1618
1661
  return;
1619
1662
  }
1620
- if (!Array.isArray(cookies)) {
1621
- usageError('Cookies file must contain a JSON array', { json: opts.json });
1622
- return;
1623
- }
1624
1663
  if (cookies.length === 0) {
1625
1664
  usageError('Cookies array must not be empty', { json: opts.json });
1626
1665
  return;
@@ -1629,7 +1668,10 @@ accountSession
1629
1668
  usageError('Each cookie must have a name (string) and value (string)', { json: opts.json });
1630
1669
  return;
1631
1670
  }
1632
- const call = () => api.put(`/api/scraps/${id}/account/session`, { cookies });
1671
+ const body = { cookies };
1672
+ if (origins)
1673
+ body.origins = origins;
1674
+ const call = () => api.put(`/api/scraps/${id}/account/session`, body);
1633
1675
  if (opts.json) {
1634
1676
  const data = await call();
1635
1677
  json(data);
@@ -1642,6 +1684,164 @@ accountSession
1642
1684
  const acc = data.account;
1643
1685
  console.log(chalk.dim(' Session: ') + (acc.hasSession ? chalk.green('✓ active') : chalk.dim('none')));
1644
1686
  });
1687
+ /**
1688
+ * @desc Map a failed `captureSession()` result onto this CLI's existing
1689
+ * exit-code taxonomy (#71/#88/#107). `no_cookies_in_scope` is the one
1690
+ * outcome where everything ran correctly but nothing was found to upload —
1691
+ * a business-level refusal (exit 1, kind:"refused"), not a usage mistake.
1692
+ *
1693
+ * `capture_failed` (trawl_cli#183 R4) is a SECOND exception, for the
1694
+ * opposite reason: launch + handshake already succeeded — Chrome was
1695
+ * reachable over CDP, a page was already open — and the failure is a
1696
+ * genuine bug in this CLI (or an unanticipated non-page-scoped failure),
1697
+ * never a "cannot run in this environment" problem. Routing it through
1698
+ * `usageError`'s exit 2 / "cannot complete in the current environment"
1699
+ * framing would misclassify it exactly the way a bare `launch_failed` used
1700
+ * to before the R4 fix. Reported instead through `reportError` with a bare
1701
+ * `Error` — `classifyError` has no dedicated `kind` for it, so it falls
1702
+ * into the generic `kind:"unknown"`/exit 1 bucket, the SAME code an
1703
+ * unhandled bug reaching index.ts's own top-level catch would already get.
1704
+ * Never `RefusalError` (this is not a business-logic refusal, everything
1705
+ * up to this point may have gone fine) and never `UsageError` (nothing
1706
+ * about the invocation or the environment was wrong).
1707
+ *
1708
+ * Every remaining reason (non_interactive, no_chrome, launch_failed, the CDP
1709
+ * pipe closing before capture, a corrupt CDP frame tearing it down) means
1710
+ * THIS invocation cannot complete in the current environment — exit 2, the
1711
+ * same bucket `confirmDestructive`'s own non-interactive refusal uses.
1712
+ */
1713
+ function reportCaptureFailure(result, opts) {
1714
+ if (result.reason === 'no_cookies_in_scope') {
1715
+ process.exitCode = reportError(new RefusalError(result.message), { json: opts.json });
1716
+ return;
1717
+ }
1718
+ if (result.reason === 'capture_failed') {
1719
+ process.exitCode = reportError(new Error(result.message), { json: opts.json });
1720
+ return;
1721
+ }
1722
+ usageError(result.message, { json: opts.json });
1723
+ }
1724
+ // account session capture — trawl_cli#183
1725
+ accountSession
1726
+ .command('capture <id>')
1727
+ .description('Open a headed Chrome to the scrap\'s target URL, wait for you to log in (2FA included), and capture the session over CDP — cookies AND localStorage. Requires a local interactive terminal with a display; does not work headless, in CI, or over a plain SSH session. You stay authenticated as yourself — Trawl never sees your credentials. Responsibility for lawful use of the captured session stays with you.')
1728
+ .option('--chrome <path>', 'Path to a Chrome/Chromium executable (auto-detected if omitted)')
1729
+ .option('--json', 'Output as JSON (counts only — a session is a bearer secret and is never printed, in any mode)')
1730
+ .action(async (id, opts) => {
1731
+ validateObjectId(id);
1732
+ // A captured session is a bearer-equivalent secret — refuse to PUT it
1733
+ // over plain HTTP (trawl_cli#183 review finding 5). Checked before
1734
+ // anything else, including the scrap lookup below (which itself
1735
+ // already sends the auth token over whatever transport is configured).
1736
+ const secureTransportError = assertSecureTransport(getApiUrl());
1737
+ if (secureTransportError) {
1738
+ usageError(secureTransportError, { json: opts.json });
1739
+ return;
1740
+ }
1741
+ const scrap = await api.get(`/api/scraps/${id}`);
1742
+ if (!scrap.url) {
1743
+ usageError(`Scrap ${id} has no target URL configured — nothing to open a browser to. A URL can be set with: trawl scraps update ${id} -u <url>`, { json: opts.json });
1744
+ return;
1745
+ }
1746
+ // Validated here, before captureSession ever creates a temp profile:
1747
+ // `existsSync` alone doesn't mean `spawn()` can run the file — a
1748
+ // non-executable path fails asynchronously deep inside chrome-launch.ts
1749
+ // instead. That failure is still caught there (an 'error' listener is
1750
+ // the safety net for anything not caught by this pre-flight — a race,
1751
+ // a permission change after this check, PUPPETEER_EXECUTABLE_PATH/
1752
+ // CHROME_PATH), but a pre-flight check gives a faster, more specific
1753
+ // message for the common case: an explicit path the user pointed at.
1754
+ if (opts.chrome) {
1755
+ const { existsSync, accessSync, constants } = await import('fs');
1756
+ if (!existsSync(opts.chrome)) {
1757
+ usageError(`Chrome not found at: ${opts.chrome}`, { json: opts.json });
1758
+ return;
1759
+ }
1760
+ try {
1761
+ accessSync(opts.chrome, constants.X_OK);
1762
+ }
1763
+ catch {
1764
+ usageError(`Chrome exists at ${opts.chrome} but is not executable.`, { json: opts.json });
1765
+ return;
1766
+ }
1767
+ }
1768
+ else if (process.env.TRAWL_CHROME_PATH) {
1769
+ // Same check, same reasoning, for the env-var form of an explicit
1770
+ // path — findChrome() gives TRAWL_CHROME_PATH top precedence and
1771
+ // would otherwise hand this same non-executable path straight to
1772
+ // captureSession.
1773
+ const trawlChromePath = process.env.TRAWL_CHROME_PATH;
1774
+ const { existsSync, accessSync, constants } = await import('fs');
1775
+ if (existsSync(trawlChromePath)) {
1776
+ try {
1777
+ accessSync(trawlChromePath, constants.X_OK);
1778
+ }
1779
+ catch {
1780
+ usageError(`Chrome exists at ${trawlChromePath} (TRAWL_CHROME_PATH) but is not executable.`, { json: opts.json });
1781
+ return;
1782
+ }
1783
+ }
1784
+ }
1785
+ // Check the same guard captureSession runs internally BEFORE printing
1786
+ // anything about a Chrome window that may never open — otherwise a
1787
+ // non-interactive run saw "a window will open" immediately followed by
1788
+ // "this cannot work headless" (trawl_cli#183 gap).
1789
+ const nonInteractiveReason = checkInteractiveEnvironment();
1790
+ if (nonInteractiveReason) {
1791
+ reportCaptureFailure({ ok: false, reason: 'non_interactive', message: nonInteractiveReason }, opts);
1792
+ return;
1793
+ }
1794
+ // Progress/prompts are stderr-only, in every mode — stdout under
1795
+ // --json must stay a single parseable document (#88/#107's rule,
1796
+ // restated for this command by trawl_cli#183's own hard rules).
1797
+ console.error(chalk.dim(`Chrome will open at ${scrap.url} for you to log in there (2FA included); capture completes when you press Enter back in this terminal.`));
1798
+ const result = await captureSession(scrap.url, opts.chrome ? { findChrome: () => opts.chrome ?? null } : {});
1799
+ if (!result.ok) {
1800
+ reportCaptureFailure(result, opts);
1801
+ return;
1802
+ }
1803
+ const call = () => api.put(`/api/scraps/${id}/account/session`, {
1804
+ cookies: result.storageState.cookies,
1805
+ origins: result.storageState.origins,
1806
+ });
1807
+ if (opts.json) {
1808
+ const data = await call();
1809
+ // Never the session itself — only what the server echoes back, plus
1810
+ // counts (#183 hard rule: no cookie/localStorage VALUE, ever, under
1811
+ // --json or otherwise).
1812
+ json({ account: data.account, targetDomain: result.targetDomain, capture: result.counts });
1813
+ return;
1814
+ }
1815
+ const data = await spin(call, {
1816
+ text: `Uploading captured session for scrap ${chalk.bold(id)}…`,
1817
+ successText: `Session captured for scrap ${chalk.bold(id)}: ${result.counts.cookiesCaptured} cookie(s), ${result.counts.originsCaptured} origin(s) in scope for ${result.targetDomain}`,
1818
+ });
1819
+ const acc = data.account;
1820
+ console.log(chalk.dim(' Session: ') + (acc.hasSession ? chalk.green('✓ active') : chalk.dim('none')));
1821
+ if (result.counts.cookiesDroppedOutOfScope > 0 || result.counts.originsDroppedOutOfScope > 0) {
1822
+ console.log(chalk.dim(` Scoped to ${result.targetDomain}: dropped ${result.counts.cookiesDroppedOutOfScope} cookie(s) and ${result.counts.originsDroppedOutOfScope} origin(s) outside that domain.`));
1823
+ }
1824
+ if (result.counts.closedEarly) {
1825
+ console.log(chalk.dim(' The browser window was closed before Enter — localStorage was not captured (cookies only).'));
1826
+ }
1827
+ if (result.counts.originsUnreadable > 0) {
1828
+ // Impossible to overlook (chalk.yellow + ⚠, not the routine chalk.dim
1829
+ // scope-drop note above) — this is a DEGRADED capture: cookies still
1830
+ // uploaded, but at least one in-scope origin's localStorage did not.
1831
+ // Two distinct causes share this one count: the page threw reading
1832
+ // `window.localStorage` (e.g. a SecurityError on partitioned
1833
+ // storage), or the read never got a response within
1834
+ // LOCALSTORAGE_READ_TIMEOUT_MS (the page's renderer was blocked — a
1835
+ // native dialog, a synchronous script, a paused debugger;
1836
+ // trawl_cli#183 review finding 3's own pipe-level timeout fix
1837
+ // reopened finding 4's exact bug class on this one path, since a
1838
+ // timeout does not close the pipe the way a corrupt frame does). The
1839
+ // two are not split apart here: both mean the same actionable fact
1840
+ // ("re-run once nothing is blocking that tab") and neither one names
1841
+ // the page URL/title/exception text — a page controls that content.
1842
+ console.log(chalk.yellow(` ⚠ ${result.counts.originsUnreadable} origin(s) could not be captured — threw reading localStorage, or gave no response within ${LOCALSTORAGE_READ_TIMEOUT_MS / 1000}s (a blocked tab). Not the same as being empty; cookies were still captured.`));
1843
+ }
1844
+ });
1645
1845
  // account status
1646
1846
  account
1647
1847
  .command('status <id>')
@@ -0,0 +1,103 @@
1
+ /**
2
+ * Raw Chrome DevTools Protocol client over Chrome's `--remote-debugging-pipe`
3
+ * transport (trawl_cli#183) — zero new dependencies. This is the SAME
4
+ * transport Puppeteer's own `pipe: true` mode uses: Chrome reads NUL
5
+ * (`\0`)-delimited JSON command messages on fd 3 and writes NUL-delimited
6
+ * JSON response/event messages on fd 4. No port, no `ws`, no HTTP polling
7
+ * of a `/json/version` endpoint, no parsing Chrome's stderr for a
8
+ * `DevTools listening on ws://…` line.
9
+ *
10
+ * Deliberately generic over any `Writable`/`Readable` pair (not tied to a
11
+ * real ChildProcess) so the framing/dispatch logic is fully testable with
12
+ * in-memory streams standing in for "Chrome" — see cdp-pipe.test.ts.
13
+ */
14
+ import type { Readable, Writable } from 'node:stream';
15
+ type EventHandler = (params: unknown, sessionId?: string) => void;
16
+ /**
17
+ * A frame Chrome sent could not be parsed as JSON (trawl_cli#183 review
18
+ * finding 3). Distinguished from the generic "Chrome closed the CDP pipe"
19
+ * error so a caller can tell "the peer sent garbage and was cut off" apart
20
+ * from "the process actually exited" — `captureSession` reports the two
21
+ * with different, factual reasons rather than defaulting a corrupt-frame
22
+ * case to "Chrome exited".
23
+ */
24
+ export declare class CdpProtocolError extends Error {
25
+ constructor(message: string);
26
+ }
27
+ /**
28
+ * A `send()` command's own deadline elapsed with no response (trawl_cli#183
29
+ * review finding 3's fix; the gap it reopened is finding 4's bug class —
30
+ * see session-capture.ts's `readCaptureDefault`). Deliberately a SIBLING of
31
+ * `CdpProtocolError`, never a subclass: a protocol error means the pipe
32
+ * itself can no longer be trusted, so it is torn down (`isClosed` flips
33
+ * true, every OTHER pending command rejects too). A timeout means exactly
34
+ * ONE command never got an answer — Chrome is still running, the pipe is
35
+ * still open, every other in-flight/future command is unaffected. Making
36
+ * this a `CdpProtocolError` subclass would let a caller's
37
+ * `instanceof CdpProtocolError` (or a future one) treat a single slow
38
+ * command as proof the whole browser is gone, which is false. Carries only
39
+ * `method` and `timeoutMs` — never `params`, which can carry page-
40
+ * controlled data.
41
+ */
42
+ export declare class CdpTimeoutError extends Error {
43
+ readonly method: string;
44
+ readonly timeoutMs: number;
45
+ constructor(method: string, timeoutMs: number);
46
+ }
47
+ export declare class CdpPipe {
48
+ private readonly writeStream;
49
+ private readonly commandTimeoutMs;
50
+ private nextId;
51
+ private buffer;
52
+ private readonly pending;
53
+ private readonly listeners;
54
+ private closed;
55
+ private protocolErr;
56
+ constructor(writeStream: Writable, readStream: Readable, commandTimeoutMs?: number);
57
+ private onClosed;
58
+ /**
59
+ * @desc A frame that fails to parse as JSON used to be dropped silently
60
+ * (`msg = undefined`, dispatch skipped) — any `send()` whose response was
61
+ * that exact frame then hung forever, since nothing ever rejected it.
62
+ * Now the whole pipe is torn down the same way a real Chrome exit is:
63
+ * every pending command rejects (with a `CdpProtocolError`, not the
64
+ * generic close message, so the cause is distinguishable), `isClosed`
65
+ * flips true, and `onClose` subscribers fire — the peer sent a frame
66
+ * that doesn't fit the protocol, so nothing it says next can be trusted
67
+ * either. The raw frame content is never included in the error message —
68
+ * it can carry page-controlled data (e.g. a `Runtime.evaluate` result).
69
+ */
70
+ private onProtocolError;
71
+ private onData;
72
+ private dispatch;
73
+ /**
74
+ * Send a CDP command and resolve with its `result`. Rejects if the pipe
75
+ * closes (Chrome exited, or a corrupt frame tore it down — see
76
+ * `onProtocolError`) before a response arrives, or with a
77
+ * `CdpTimeoutError` if no response arrives within `timeoutMs` (defaults
78
+ * to the pipe's own `commandTimeoutMs`, 30s) — the timeout names the
79
+ * METHOD, never `params` (which can carry page-controlled data).
80
+ * @param timeoutMs — per-call override. Lets a caller pin its own named
81
+ * deadline as an assertable constant rather than relying on the pipe's
82
+ * default silently matching (see session-capture.ts's
83
+ * `LOCALSTORAGE_READ_TIMEOUT_MS`, which currently equals this pipe's own
84
+ * 30s default but is passed explicitly anyway — a post-cap review found
85
+ * that giving `Runtime.evaluate` a SHORTER override here, on the
86
+ * assumption a page-blocked renderer "won't unblock itself in 30s any
87
+ * more than in 5", cut off real-but-slow work that finished on its own;
88
+ * there is no override value proven safe, so this parameter exists for
89
+ * callers that want one, not as a recommendation to use a shorter one).
90
+ */
91
+ send<T = unknown>(method: string, params?: unknown, sessionId?: string, timeoutMs?: number): Promise<T>;
92
+ /** Subscribe to a CDP event (`'Target.targetDestroyed'`, …). Returns an
93
+ * unsubscribe function. */
94
+ on(method: string, handler: EventHandler): () => void;
95
+ /** Subscribe to the pipe closing (Chrome process gone). */
96
+ onClose(handler: () => void): () => void;
97
+ get isClosed(): boolean;
98
+ /** Set only when the pipe was torn down because of a corrupt frame — as
99
+ * opposed to a real Chrome exit — so a caller can report the actual
100
+ * cause instead of defaulting every close to "Chrome exited". */
101
+ get protocolError(): CdpProtocolError | undefined;
102
+ }
103
+ export {};
@@ -0,0 +1,221 @@
1
+ /** Reserved internal event name — never collides with a real CDP method
2
+ * (every real one is `Domain.method`, always containing a dot). Emitted
3
+ * once the read side of the pipe ends/errors, so any in-flight capture
4
+ * can tell "Chrome went away" apart from "the browser answered". */
5
+ const PIPE_CLOSED = '__pipe_closed__';
6
+ /** A per-command deadline that never fires under normal operation. */
7
+ const DEFAULT_COMMAND_TIMEOUT_MS = 30_000;
8
+ /**
9
+ * A frame Chrome sent could not be parsed as JSON (trawl_cli#183 review
10
+ * finding 3). Distinguished from the generic "Chrome closed the CDP pipe"
11
+ * error so a caller can tell "the peer sent garbage and was cut off" apart
12
+ * from "the process actually exited" — `captureSession` reports the two
13
+ * with different, factual reasons rather than defaulting a corrupt-frame
14
+ * case to "Chrome exited".
15
+ */
16
+ export class CdpProtocolError extends Error {
17
+ constructor(message) {
18
+ super(message);
19
+ this.name = 'CdpProtocolError';
20
+ }
21
+ }
22
+ /**
23
+ * A `send()` command's own deadline elapsed with no response (trawl_cli#183
24
+ * review finding 3's fix; the gap it reopened is finding 4's bug class —
25
+ * see session-capture.ts's `readCaptureDefault`). Deliberately a SIBLING of
26
+ * `CdpProtocolError`, never a subclass: a protocol error means the pipe
27
+ * itself can no longer be trusted, so it is torn down (`isClosed` flips
28
+ * true, every OTHER pending command rejects too). A timeout means exactly
29
+ * ONE command never got an answer — Chrome is still running, the pipe is
30
+ * still open, every other in-flight/future command is unaffected. Making
31
+ * this a `CdpProtocolError` subclass would let a caller's
32
+ * `instanceof CdpProtocolError` (or a future one) treat a single slow
33
+ * command as proof the whole browser is gone, which is false. Carries only
34
+ * `method` and `timeoutMs` — never `params`, which can carry page-
35
+ * controlled data.
36
+ */
37
+ export class CdpTimeoutError extends Error {
38
+ method;
39
+ timeoutMs;
40
+ constructor(method, timeoutMs) {
41
+ super(`CDP command '${method}' timed out after ${timeoutMs}ms without a response`);
42
+ this.method = method;
43
+ this.timeoutMs = timeoutMs;
44
+ this.name = 'CdpTimeoutError';
45
+ }
46
+ }
47
+ export class CdpPipe {
48
+ writeStream;
49
+ commandTimeoutMs;
50
+ nextId = 1;
51
+ buffer = '';
52
+ pending = new Map();
53
+ listeners = new Map();
54
+ closed = false;
55
+ protocolErr;
56
+ constructor(writeStream, readStream, commandTimeoutMs = DEFAULT_COMMAND_TIMEOUT_MS) {
57
+ this.writeStream = writeStream;
58
+ this.commandTimeoutMs = commandTimeoutMs;
59
+ readStream.setEncoding('utf8');
60
+ readStream.on('data', (chunk) => this.onData(chunk));
61
+ readStream.on('end', () => this.onClosed());
62
+ readStream.on('close', () => this.onClosed());
63
+ readStream.on('error', () => this.onClosed());
64
+ // fd 3 (write) and fd 4 (read) are separate sockets — Chrome dying can
65
+ // surface as an EPIPE on a write that lands before the read side's
66
+ // 'end'/'close' ever fires. An unhandled 'error' on a Writable is a
67
+ // thrown exception with no listener, which would kill the whole CLI
68
+ // process and skip captureSession's `finally` (so the temp profile is
69
+ // never cleaned up) — routing it through the same onClosed() path
70
+ // keeps "Chrome went away" a single, always-handled state.
71
+ writeStream.on('error', () => this.onClosed());
72
+ }
73
+ onClosed(err) {
74
+ if (this.closed)
75
+ return;
76
+ this.closed = true;
77
+ const closeErr = err ?? new Error('Chrome closed the CDP pipe');
78
+ for (const { reject } of this.pending.values())
79
+ reject(closeErr);
80
+ this.pending.clear();
81
+ const set = this.listeners.get(PIPE_CLOSED);
82
+ if (set)
83
+ for (const fn of set)
84
+ fn(undefined);
85
+ }
86
+ /**
87
+ * @desc A frame that fails to parse as JSON used to be dropped silently
88
+ * (`msg = undefined`, dispatch skipped) — any `send()` whose response was
89
+ * that exact frame then hung forever, since nothing ever rejected it.
90
+ * Now the whole pipe is torn down the same way a real Chrome exit is:
91
+ * every pending command rejects (with a `CdpProtocolError`, not the
92
+ * generic close message, so the cause is distinguishable), `isClosed`
93
+ * flips true, and `onClose` subscribers fire — the peer sent a frame
94
+ * that doesn't fit the protocol, so nothing it says next can be trusted
95
+ * either. The raw frame content is never included in the error message —
96
+ * it can carry page-controlled data (e.g. a `Runtime.evaluate` result).
97
+ */
98
+ onProtocolError() {
99
+ if (this.closed)
100
+ return;
101
+ const err = new CdpProtocolError('Chrome sent a CDP frame that failed to parse as JSON — the pipe can no longer be trusted and has been torn down; any in-flight command was rejected.');
102
+ this.protocolErr = err;
103
+ this.onClosed(err);
104
+ }
105
+ onData(chunk) {
106
+ // Once torn down (a real close, or a corrupt frame — onProtocolError)
107
+ // the read stream can still be alive and keep delivering chunks from
108
+ // an untrusted peer; without this, a later VALID-looking event frame
109
+ // would still `dispatch()` to any listener still registered instead
110
+ // of being ignored, and `this.buffer` would keep growing forever.
111
+ if (this.closed)
112
+ return;
113
+ this.buffer += chunk;
114
+ let idx = this.buffer.indexOf('\0');
115
+ while (idx !== -1) {
116
+ const raw = this.buffer.slice(0, idx);
117
+ this.buffer = this.buffer.slice(idx + 1);
118
+ if (raw.length > 0) {
119
+ let msg;
120
+ try {
121
+ msg = JSON.parse(raw);
122
+ }
123
+ catch {
124
+ this.onProtocolError();
125
+ return; // torn down — whatever else is buffered is moot
126
+ }
127
+ this.dispatch(msg);
128
+ }
129
+ idx = this.buffer.indexOf('\0');
130
+ }
131
+ }
132
+ dispatch(msg) {
133
+ if (typeof msg.id === 'number' && this.pending.has(msg.id)) {
134
+ const entry = this.pending.get(msg.id);
135
+ this.pending.delete(msg.id);
136
+ if (msg.error)
137
+ entry.reject(new Error(`${msg.error.message} (code ${msg.error.code})`));
138
+ else
139
+ entry.resolve(msg.result);
140
+ return;
141
+ }
142
+ if (typeof msg.method === 'string') {
143
+ const set = this.listeners.get(msg.method);
144
+ if (set)
145
+ for (const fn of set)
146
+ fn(msg.params, msg.sessionId);
147
+ }
148
+ }
149
+ /**
150
+ * Send a CDP command and resolve with its `result`. Rejects if the pipe
151
+ * closes (Chrome exited, or a corrupt frame tore it down — see
152
+ * `onProtocolError`) before a response arrives, or with a
153
+ * `CdpTimeoutError` if no response arrives within `timeoutMs` (defaults
154
+ * to the pipe's own `commandTimeoutMs`, 30s) — the timeout names the
155
+ * METHOD, never `params` (which can carry page-controlled data).
156
+ * @param timeoutMs — per-call override. Lets a caller pin its own named
157
+ * deadline as an assertable constant rather than relying on the pipe's
158
+ * default silently matching (see session-capture.ts's
159
+ * `LOCALSTORAGE_READ_TIMEOUT_MS`, which currently equals this pipe's own
160
+ * 30s default but is passed explicitly anyway — a post-cap review found
161
+ * that giving `Runtime.evaluate` a SHORTER override here, on the
162
+ * assumption a page-blocked renderer "won't unblock itself in 30s any
163
+ * more than in 5", cut off real-but-slow work that finished on its own;
164
+ * there is no override value proven safe, so this parameter exists for
165
+ * callers that want one, not as a recommendation to use a shorter one).
166
+ */
167
+ send(method, params = {}, sessionId, timeoutMs = this.commandTimeoutMs) {
168
+ if (this.closed)
169
+ return Promise.reject(this.protocolErr ?? new Error('Chrome closed the CDP pipe'));
170
+ const id = this.nextId++;
171
+ const req = { id, method, params };
172
+ if (sessionId)
173
+ req.sessionId = sessionId;
174
+ return new Promise((resolve, reject) => {
175
+ // Every settle path (dispatch's resolve/reject, this timeout, a
176
+ // write-side EPIPE below, and onClosed's reject-all on a pipe
177
+ // teardown) goes through these two wrappers, so the timer is always
178
+ // cleared exactly once no matter which path wins the race.
179
+ const settleResolve = (v) => {
180
+ clearTimeout(timer);
181
+ resolve(v);
182
+ };
183
+ const settleReject = (e) => {
184
+ clearTimeout(timer);
185
+ reject(e);
186
+ };
187
+ const timer = setTimeout(() => {
188
+ this.pending.delete(id);
189
+ settleReject(new CdpTimeoutError(method, timeoutMs));
190
+ }, timeoutMs);
191
+ this.pending.set(id, { resolve: settleResolve, reject: settleReject });
192
+ this.writeStream.write(`${JSON.stringify(req)}\0`, (err) => {
193
+ if (err) {
194
+ this.pending.delete(id);
195
+ settleReject(err);
196
+ }
197
+ });
198
+ });
199
+ }
200
+ /** Subscribe to a CDP event (`'Target.targetDestroyed'`, …). Returns an
201
+ * unsubscribe function. */
202
+ on(method, handler) {
203
+ if (!this.listeners.has(method))
204
+ this.listeners.set(method, new Set());
205
+ this.listeners.get(method).add(handler);
206
+ return () => this.listeners.get(method)?.delete(handler);
207
+ }
208
+ /** Subscribe to the pipe closing (Chrome process gone). */
209
+ onClose(handler) {
210
+ return this.on(PIPE_CLOSED, () => handler());
211
+ }
212
+ get isClosed() {
213
+ return this.closed;
214
+ }
215
+ /** Set only when the pipe was torn down because of a corrupt frame — as
216
+ * opposed to a real Chrome exit — so a caller can report the actual
217
+ * cause instead of defaulting every close to "Chrome exited". */
218
+ get protocolError() {
219
+ return this.protocolErr;
220
+ }
221
+ }
@@ -0,0 +1,12 @@
1
+ /**
2
+ * @desc Resolve the Chrome executable to launch. Precedence: an explicit
3
+ * `--chrome <path>` flag (validated by the caller) > `TRAWL_CHROME_PATH` >
4
+ * `PUPPETEER_EXECUTABLE_PATH`/`CHROME_PATH` (common conventions other
5
+ * tools already set) > the well-known per-OS install paths. Returns null
6
+ * when nothing is found — the caller is responsible for a factual,
7
+ * non-guessing error message.
8
+ * @param {NodeJS.ProcessEnv} env
9
+ * @param {NodeJS.Platform} platform
10
+ * @param {(path: string) => boolean} exists — injectable for tests.
11
+ */
12
+ export declare function findChrome(env?: NodeJS.ProcessEnv, platform?: NodeJS.Platform, exists?: (path: string) => boolean): string | null;