@trawlme/cli 3.11.0 → 3.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,600 @@
1
+ /**
2
+ * Orchestrates `trawl scraps account session capture <id>` (trawl_cli#183):
3
+ * launch a headed, isolated Chrome; let a human log in (2FA included —
4
+ * Trawl/the agent never sees credentials); read the resulting session over
5
+ * CDP; scope it to the target domain; hand back a storageState-shaped
6
+ * result. Does NOT talk to the Trawl API — the command layer
7
+ * (commands/scraps.ts) owns the `PUT /api/scraps/:id/account/session`
8
+ * call, same as every other command in this CLI owns its own `api.*`
9
+ * calls.
10
+ *
11
+ * Every I/O boundary (Chrome discovery/launch, the CDP calls themselves,
12
+ * the temp profile, the "done" prompt) is behind the `Deps` interface so
13
+ * the orchestration logic — the guard, the race, cleanup-always, count
14
+ * assembly — is fully testable without a real browser. `captureSession`
15
+ * itself never throws for an EXPECTED failure (no chrome, non-interactive,
16
+ * zero in-scope cookies) — those come back as `{ ok: false, reason,
17
+ * message }` for the command layer to map to the right exit code. An
18
+ * UNEXPECTED failure (Chrome crashed mid-flight, a CDP call errored) is
19
+ * caught and returned the same way rather than propagated, so cleanup
20
+ * always still runs via the single `finally`.
21
+ */
22
+ import { mkdtempSync, rmSync } from 'node:fs';
23
+ import { tmpdir } from 'node:os';
24
+ import { join } from 'node:path';
25
+ import { createInterface } from 'node:readline';
26
+ import { findChrome } from './chrome-discovery.js';
27
+ import { launchChrome } from './chrome-launch.js';
28
+ import { detectNonInteractive } from './session-capture-guard.js';
29
+ import { mapCookies, mapOrigins, getTargetHost, isOriginHostInScope, } from './storage-state.js';
30
+ /** Shared between the two call sites that can observe Chrome having
31
+ * exited before a capture completed (mid-wait, and mid-readCapture). */
32
+ const PROCESS_EXITED_MESSAGE = 'Chrome exited before the session was read — nothing uploaded. The capture completes on Enter in the terminal; Chrome exiting first ends the run without a session.';
33
+ /** Shared between the two call sites that can observe a corrupt CDP frame
34
+ * tearing the pipe down (mid-wait, and mid-readCapture) — trawl_cli#183
35
+ * review finding 3. `protocolError.message` is internal, factual text
36
+ * with no page-controlled content (see cdp-pipe.ts's onProtocolError). */
37
+ function cdpProtocolErrorMessage(protocolError) {
38
+ return `${protocolError.message} Nothing was uploaded.`;
39
+ }
40
+ /**
41
+ * The three real (non-fake) CDP orchestration steps, exported for direct
42
+ * testing against a minimal fake CdpPipe — session-capture.test.ts covers
43
+ * the OUTER guard/race/cleanup logic with these swapped out entirely;
44
+ * session-capture-defaults.test.ts covers these directly instead, since
45
+ * "test the wiring" and "test the wired thing" are two different jobs.
46
+ */
47
+ export async function findPageTargetDefault(cdp) {
48
+ const deadline = Date.now() + 10_000;
49
+ for (;;) {
50
+ const { targetInfos } = await cdp.send('Target.getTargets');
51
+ const page = targetInfos.find((t) => t.type === 'page');
52
+ if (page)
53
+ return page.targetId;
54
+ if (Date.now() > deadline)
55
+ throw new Error('Chrome did not open a page target in time');
56
+ await new Promise((r) => setTimeout(r, 100));
57
+ }
58
+ }
59
+ export async function waitForDoneDefault(cdp, targetId, cleanup) {
60
+ await cdp.send('Target.setDiscoverTargets', { discover: true });
61
+ return new Promise((resolve) => {
62
+ let settled = false;
63
+ const offDestroyed = cdp.on('Target.targetDestroyed', (params) => {
64
+ const p = params;
65
+ if (!settled && p.targetId === targetId) {
66
+ settled = true;
67
+ offDestroyed();
68
+ offClose();
69
+ rl.close();
70
+ resolve({ closedEarly: true });
71
+ }
72
+ });
73
+ // If the whole Chrome process dies mid-wait (crash, or a platform where
74
+ // closing the last window quits it) no `targetDestroyed` event can ever
75
+ // arrive on a dead pipe — without this, the prompt sits there forever
76
+ // giving the human no feedback. `captureSession` checks its own
77
+ // `pipeClosed` flag right after this resolves and turns it into the
78
+ // factual `process_exited_before_capture` result immediately, rather
79
+ // than waiting on a keypress that was never going to fix anything.
80
+ const offClose = cdp.onClose(() => {
81
+ if (!settled) {
82
+ settled = true;
83
+ offDestroyed();
84
+ rl.close();
85
+ resolve({ closedEarly: true });
86
+ }
87
+ });
88
+ const rl = createInterface({ input: process.stdin, output: process.stderr, terminal: true });
89
+ // Ctrl-C during this prompt never reaches `process` as a real SIGINT —
90
+ // raw-mode TTY input disables signal generation for it — so the
91
+ // Interface's own synthesized 'SIGINT' is the only place to catch it.
92
+ rl.on('SIGINT', () => {
93
+ settled = true;
94
+ offDestroyed();
95
+ offClose();
96
+ rl.close();
97
+ cleanup();
98
+ process.exit(130);
99
+ });
100
+ rl.question('Capture completes on Enter once you are signed in, or when the Chrome window is closed: ', () => {
101
+ if (!settled) {
102
+ settled = true;
103
+ offDestroyed();
104
+ offClose();
105
+ rl.close();
106
+ resolve({ closedEarly: false });
107
+ }
108
+ });
109
+ });
110
+ }
111
+ /**
112
+ * @desc The deadline for the ONE CDP command in this file that runs inside
113
+ * a page's own renderer rather than Chrome's browser process:
114
+ * `Runtime.evaluate` reading `window.localStorage`. Deliberately the SAME
115
+ * 30s as `CdpPipe`'s own browser-process default, not a shorter override —
116
+ * a post-cap review (trawl_cli#183) found that an earlier 5s value here cut
117
+ * off a real, terminating synchronous computation (the shape of an
118
+ * anti-bot challenge solve) at ~5s: `originsUnreadable:1` for a page that
119
+ * would have captured cleanly at 7s. There is no value that is provably
120
+ * "long enough" — a renderer truly blocked forever (a native dialog, a
121
+ * paused debugger) costs the same whether the ceiling is 5s or 30s, while
122
+ * a renderer doing bounded work of unknown-but-finite length keeps a
123
+ * chance of finishing for as long as this stays generous. 30s is a
124
+ * JUDGEMENT call on that trade-off, not a provably-correct number — kept
125
+ * equal to the pipe's own default so this file doesn't invent a second
126
+ * number to defend. Passed explicitly to `send()` anyway (rather than
127
+ * relying on the pipe's own default silently matching) so the deadline
128
+ * stays a named, assertable constant here regardless of what the pipe
129
+ * default happens to be. Whatever the outcome, it is always COUNTED and
130
+ * SURFACED (`originsUnreadable`, plus the human-facing ⚠ line in
131
+ * scraps.ts) — never silently swallowed; see `readCaptureDefault` below
132
+ * for the stderr progress line that covers the wait itself.
133
+ */
134
+ export const LOCALSTORAGE_READ_TIMEOUT_MS = 30_000;
135
+ /** How long to wait, mid-read, before printing a factual "still waiting"
136
+ * progress line to stderr — a human who just pressed Enter and is now
137
+ * staring at silence for up to `LOCALSTORAGE_READ_TIMEOUT_MS` has no way
138
+ * to tell "still working" from "hung". Well under the read deadline so it
139
+ * fires long before the read could time out; never printed for a read
140
+ * that resolves before this fires (cleared in a `finally`). */
141
+ const LOCALSTORAGE_READ_PROGRESS_MS = 5_000;
142
+ /**
143
+ * @desc Structural guard for `Runtime.evaluate`'s returned value (see
144
+ * `readCaptureDefault`'s call site) — trawl_cli#183 post-cap review finding
145
+ * B. `returnByValue: true` means CDP itself, not page script, produced this
146
+ * value, but a page can still make the EXPRESSION return arbitrary garbage
147
+ * without throwing (tampering with `Object.entries`, `Array.prototype.map`,
148
+ * etc.) — this is the backstop for that, not a claim that the expression
149
+ * itself is un-tamperable. Anything that fails this check is treated as
150
+ * page-controlled misbehavior, counted the same as a thrown read.
151
+ */
152
+ function isLocalStorageEntriesShape(value) {
153
+ return (Array.isArray(value) &&
154
+ value.every((e) => typeof e === 'object' && e !== null && typeof e.name === 'string' && typeof e.value === 'string'));
155
+ }
156
+ /**
157
+ * @desc Phase 1 of the per-page read (trawl_cli#183 R4 fix): every CDP
158
+ * round-trip SCOPED TO ONE PAGE — attach, the localStorage evaluate,
159
+ * detach. Every round this catch block went through classified errors by
160
+ * TYPE (a `CdpTimeoutError` counts, "anything else" rethrows) — the wrong
161
+ * criterion, because a plain `Error` is exactly what BOTH a page-caused CDP
162
+ * rejection and a CLI bug look like; no error-class taxonomy tells them
163
+ * apart. The right discriminant is PROVENANCE: did the rejection come out
164
+ * of a CDP call for this page, or out of this file's own code running on
165
+ * data CDP already handed back? This function's OWN BOUNDARY draws that
166
+ * line structurally instead — `readCaptureDefault` treats ANY rejection
167
+ * out of here as page-caused, whatever its error class: a `CdpTimeoutError`
168
+ * (the renderer never answered — a native dialog, a synchronous script, a
169
+ * paused debugger), or a plain rejection because the target closed
170
+ * mid-attach/mid-evaluate (reproduced against real Chrome: a second
171
+ * in-scope page — an OAuth popup, say — closing at the moment the loop
172
+ * reaches it rejects `Target.attachToTarget` with a plain
173
+ * `Error("No target with given id found (code -32602)")`, pipe still open,
174
+ * not a timeout). The one exception `readCaptureDefault` still special-
175
+ * cases on the way out is `cdp.isClosed` — that means the PIPE itself
176
+ * died, not this page, and belongs to every other page too.
177
+ */
178
+ async function readPageOverCdp(cdp, targetId) {
179
+ let sessionId;
180
+ try {
181
+ const attached = await cdp.send('Target.attachToTarget', {
182
+ targetId,
183
+ flatten: true,
184
+ });
185
+ sessionId = attached.sessionId;
186
+ // A human who just pressed Enter has no signal at all while this one
187
+ // call is in flight — print a factual "still waiting" line to stderr
188
+ // if it runs past LOCALSTORAGE_READ_PROGRESS_MS, so a bounded-but-slow
189
+ // page (or a genuinely blocked one) doesn't read as a hung CLI.
190
+ // Cleared unconditionally in the inner `finally` so it never fires
191
+ // for a read that already resolved. Never the page's URL/title —
192
+ // only the fact that a read is in progress.
193
+ const progressTimer = setTimeout(() => {
194
+ process.stderr.write(` still waiting for a page to finish its scripts (up to ${LOCALSTORAGE_READ_TIMEOUT_MS / 1000}s)…\n`);
195
+ }, LOCALSTORAGE_READ_PROGRESS_MS);
196
+ try {
197
+ return await cdp.send('Runtime.evaluate', {
198
+ // Builds the {name,value}[] array directly and returns it via
199
+ // CDP's OWN serialisation (`returnByValue: true`) rather than
200
+ // calling the page's `JSON.stringify` and parsing the result
201
+ // ourselves — trawl_cli#183 post-cap review finding B: a page
202
+ // that overrides `window.JSON.stringify` to return `"null"`
203
+ // made our own later `JSON.parse` + `.length` throw a TypeError
204
+ // that used to be laundered into `originsUnreadable`,
205
+ // indistinguishable from an ordinary blocked-tab capture.
206
+ // Chrome's inspector serialiser cannot be tampered with from
207
+ // page script the way `JSON.stringify` can.
208
+ expression: 'Object.entries(window.localStorage).map(([name, value]) => ({ name, value }))',
209
+ returnByValue: true,
210
+ }, sessionId, LOCALSTORAGE_READ_TIMEOUT_MS);
211
+ }
212
+ finally {
213
+ clearTimeout(progressTimer);
214
+ }
215
+ }
216
+ finally {
217
+ if (sessionId) {
218
+ try {
219
+ await cdp.send('Target.detachFromTarget', { sessionId });
220
+ }
221
+ catch {
222
+ // already detached / target gone — non-fatal either way: this
223
+ // page's origin is already about to be counted or captured
224
+ }
225
+ }
226
+ }
227
+ }
228
+ export async function readCaptureDefault(cdp, targetUrl, closedEarly) {
229
+ const { cookies } = await cdp.send('Storage.getCookies', {});
230
+ if (closedEarly)
231
+ return { cookies, origins: [], originsUnreadable: 0 };
232
+ const targetHost = getTargetHost(targetUrl);
233
+ const { targetInfos } = await cdp.send('Target.getTargets');
234
+ const pages = targetInfos.filter((t) => t.type === 'page');
235
+ const origins = [];
236
+ let originsUnreadable = 0;
237
+ for (const page of pages) {
238
+ let origin;
239
+ try {
240
+ const u = new URL(page.url);
241
+ if (!isOriginHostInScope(u.hostname, targetHost))
242
+ continue;
243
+ origin = u.origin;
244
+ }
245
+ catch {
246
+ continue;
247
+ }
248
+ // Phase 1 — readPageOverCdp above; see its own doc comment. Any
249
+ // rejection out of it is page-caused BY PROVENANCE, not by error class
250
+ // (trawl_cli#183 R4) — count it and move on to the next page. The one
251
+ // exception is `cdp.isClosed`: if the WHOLE pipe was torn down (Chrome
252
+ // exited, or a corrupt frame — trawl_cli#183 review finding 3), every
253
+ // remaining `cdp.send` in this loop would reject the same way;
254
+ // swallowing that here would finish the loop and report an apparently-
255
+ // successful capture with fewer origins than real. Rethrown so
256
+ // captureSession's own pipeClosed/protocolError handling reports the
257
+ // actual cause. Checked unconditionally, before looking at the
258
+ // rejection at all — a dead pipe is true regardless of which error
259
+ // under it happens to be surfacing, including a `CdpTimeoutError` that
260
+ // raced a teardown on the same tick (a `CdpProtocolError` teardown
261
+ // flips `isClosed` before it rejects any pending command — see
262
+ // cdp-pipe.ts's `onClosed` — so this one check already covers both
263
+ // causes; there is no separate `instanceof CdpProtocolError` branch to
264
+ // maintain).
265
+ let evalResult;
266
+ try {
267
+ evalResult = await readPageOverCdp(cdp, page.targetId);
268
+ }
269
+ catch (e) {
270
+ if (cdp.isClosed)
271
+ throw e;
272
+ originsUnreadable++;
273
+ process.stderr.write(' one in-scope page could not be read over CDP — counted, continuing…\n');
274
+ continue;
275
+ }
276
+ // Phase 2 — post-processing: a decision THIS FILE's OWN code makes on
277
+ // data CDP already handed back, never another CDP call. Deliberately
278
+ // OUTSIDE the try/catch above (trawl_cli#183 R4 fix): a bug in this
279
+ // code must propagate as a genuinely FAILED capture, never get caught
280
+ // by the page-provenance catch above and laundered into one more
281
+ // `originsUnreadable` count indistinguishable from an ordinary blocked
282
+ // tab. That laundering is exactly the bug class R3 fixed once already
283
+ // (trawl_cli#183 post-cap review finding B) — classifying by error TYPE
284
+ // inside a single catch cannot keep it fixed, because a plain `Error`
285
+ // is exactly what BOTH a page-caused CDP rejection (R4) and a CLI bug
286
+ // (R3) look like; a FUNCTION BOUNDARY can, because Phase 2 never runs
287
+ // inside Phase 1's `try`. Page garbage is still handled here, by the
288
+ // structural `isLocalStorageEntriesShape` guard below — but that is a
289
+ // decision this code makes on data, not an exception this code merely
290
+ // swallows.
291
+ if (evalResult.exceptionDetails) {
292
+ // Runtime.evaluate SUCCEEDED at the CDP layer but the expression
293
+ // threw INSIDE the page (e.g. a SecurityError on partitioned/
294
+ // sandboxed storage) — result.value is then undefined, which used
295
+ // to be silently read as "0 keys" and the origin dropped, making a
296
+ // blocked read indistinguishable from genuinely empty localStorage
297
+ // (trawl_cli#183 review finding 4). Counted separately; never the
298
+ // exception text itself — a page controls that string.
299
+ originsUnreadable++;
300
+ continue;
301
+ }
302
+ const value = evalResult.result?.value;
303
+ if (!isLocalStorageEntriesShape(value)) {
304
+ // Runtime.evaluate raised no page-side exception and CDP's own
305
+ // serialiser did its job, but the returned value is not the
306
+ // `{name,value}[]` shape the expression asked for — a page can
307
+ // still tamper with a global to hand back arbitrary garbage WITHOUT
308
+ // ever throwing (trawl_cli#183 post-cap review finding B). That is
309
+ // page-controlled misbehavior, exactly the same actionable fact as a
310
+ // thrown read — count it, never trust the shape blindly, and never
311
+ // surface its content (a page controls it).
312
+ originsUnreadable++;
313
+ continue;
314
+ }
315
+ if (value.length > 0) {
316
+ origins.push({ origin, entries: value });
317
+ }
318
+ }
319
+ return { cookies, origins, originsUnreadable };
320
+ }
321
+ /**
322
+ * @desc `rmSync` with a short, bounded, SYNCHRONOUS retry (max ~200ms
323
+ * total). Verified against a real Chrome launch (trawl_cli#183's loopback
324
+ * E2E): even after killing Chrome's whole process group (see
325
+ * chrome-launch.ts's `detached`), a helper subprocess can hold a file
326
+ * under the profile open for a few milliseconds after the kill signal is
327
+ * delivered — a bare `rmSync` right after `kill()` lost that race and
328
+ * silently leaked the temp profile every time. Stays synchronous
329
+ * (`Atomics.wait`, no `await`) because cleanup must be callable from a
330
+ * signal handler right before `process.exit()`.
331
+ */
332
+ export function rmSyncWithRetry(dir) {
333
+ const MAX_ATTEMPTS = 5;
334
+ for (let attempt = 1; attempt <= MAX_ATTEMPTS; attempt++) {
335
+ try {
336
+ rmSync(dir, { recursive: true, force: true });
337
+ return;
338
+ }
339
+ catch (e) {
340
+ if (attempt === MAX_ATTEMPTS)
341
+ throw e;
342
+ Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, 40);
343
+ }
344
+ }
345
+ }
346
+ /** The conventional shell exit code for each terminating signal this
347
+ * process still handles (128 + signal number) — reused here so a killed
348
+ * capture reports the same code a bare, unhandled kill would have. */
349
+ const TERMINATION_EXIT_CODES = {
350
+ SIGHUP: 129,
351
+ SIGINT: 130,
352
+ SIGTERM: 143,
353
+ };
354
+ /**
355
+ * @desc The default `onSigint` implementation, exported for direct testing
356
+ * (same pattern as `findPageTargetDefault`/`waitForDoneDefault`/
357
+ * `readCaptureDefault` above — this file's own convention keeps "test the
358
+ * wiring" and "test the wired thing" separate). Registers `cleanup` on
359
+ * SIGINT, SIGTERM, AND SIGHUP (trawl_cli#183 review finding 1 — only
360
+ * SIGINT was registered before this; `timeout` without `--signal`, a bare
361
+ * `kill <pid>`, most supervisors, and closing the terminal tab mid-login
362
+ * all default-terminate a process via SIGTERM/SIGHUP, and Node runs no
363
+ * `finally` for a default-terminated process). SIGKILL is the one signal
364
+ * this — or any handler, in any process — cannot observe: the OS tears
365
+ * the process down directly, no userspace code runs at all, so Chrome and
366
+ * the temp profile are left behind whenever that specific signal is what
367
+ * ends this process. That gap is inherent, not something a different
368
+ * signal list here could close.
369
+ */
370
+ export function registerTerminationHandlers(cleanup) {
371
+ const signals = Object.keys(TERMINATION_EXIT_CODES);
372
+ const handlers = signals.map((signal) => {
373
+ const handler = () => {
374
+ cleanup();
375
+ process.exit(TERMINATION_EXIT_CODES[signal]);
376
+ };
377
+ process.on(signal, handler);
378
+ return { signal, handler };
379
+ });
380
+ return () => {
381
+ for (const { signal, handler } of handlers)
382
+ process.off(signal, handler);
383
+ };
384
+ }
385
+ /**
386
+ * @desc The same guard `captureSession` runs internally as its very first
387
+ * step, exposed so the command layer can call it BEFORE printing anything
388
+ * about a Chrome window that may never open. Without this, `scraps.ts`
389
+ * printed its "a Chrome window will open" banner unconditionally, ahead of
390
+ * this exact check inside `captureSession` — so a non-interactive run saw
391
+ * that banner immediately followed by "this cannot work headless"
392
+ * (trawl_cli#183 gap). Reads live process state once; pure
393
+ * `detectNonInteractive` stays the single source of truth for both call
394
+ * sites.
395
+ * @returns {string | null} a factual reason when this command cannot
396
+ * proceed in the current environment, or null when it can.
397
+ */
398
+ export function checkInteractiveEnvironment() {
399
+ return detectNonInteractive({ stdinIsTTY: Boolean(process.stdin.isTTY), platform: process.platform, env: process.env });
400
+ }
401
+ function defaultDeps() {
402
+ return {
403
+ detectNonInteractive: checkInteractiveEnvironment,
404
+ findChrome: () => findChrome(),
405
+ mkdtemp: () => mkdtempSync(join(tmpdir(), 'trawl-session-')),
406
+ rmSync: rmSyncWithRetry,
407
+ launch: (chromePath, targetUrl, userDataDir) => launchChrome(chromePath, targetUrl, userDataDir),
408
+ findPageTarget: findPageTargetDefault,
409
+ waitForDone: waitForDoneDefault,
410
+ readCapture: readCaptureDefault,
411
+ onSigint: registerTerminationHandlers,
412
+ };
413
+ }
414
+ /**
415
+ * @desc Run the full capture flow for one scrap's target URL.
416
+ * @param {string} targetUrl
417
+ * @param {Partial<CaptureDeps>} depsOverride — for tests; real deps fill in
418
+ * the rest.
419
+ * @returns {Promise<CaptureResult>}
420
+ */
421
+ export async function captureSession(targetUrl, depsOverride = {}) {
422
+ const deps = { ...defaultDeps(), ...depsOverride };
423
+ const guardMessage = deps.detectNonInteractive();
424
+ if (guardMessage)
425
+ return { ok: false, reason: 'non_interactive', message: guardMessage };
426
+ const chromePath = deps.findChrome();
427
+ if (!chromePath) {
428
+ return {
429
+ ok: false,
430
+ reason: 'no_chrome',
431
+ message: 'No Chrome or Chromium executable was found. Google Chrome, once installed, is auto-detected; TRAWL_CHROME_PATH names one explicitly.',
432
+ };
433
+ }
434
+ const userDataDir = deps.mkdtemp();
435
+ let proc;
436
+ let cleaned = false;
437
+ // Covers every path this process can observe: return, throw, and SIGINT,
438
+ // SIGTERM, and SIGHUP (via onSigint/registerTerminationHandlers above —
439
+ // trawl_cli#183 review finding 1: SIGTERM/SIGHUP were missing before
440
+ // this, so `timeout` without `--signal`, a bare `kill <pid>`, most
441
+ // supervisors, and closing the terminal tab mid-login all skipped
442
+ // cleanup). A SIGKILL of this process itself (OOM, a CI timeout,
443
+ // `kill -9`) is NOT observable by any handler here — Chrome exits on its
444
+ // own when the CDP pipe hits EOF, but this cleanup, and so the temp
445
+ // profile removal below, never runs. That gap is inherent to any
446
+ // local-Chrome-launch design (not specific to the pipe transport) and is
447
+ // not closed here.
448
+ const cleanup = () => {
449
+ if (cleaned)
450
+ return;
451
+ cleaned = true;
452
+ try {
453
+ if (proc?.pid) {
454
+ // Kill Chrome's WHOLE process group (it and its own renderer/GPU/
455
+ // network-service helpers — chrome-launch.ts launches it
456
+ // `detached` on POSIX for exactly this), never just the single
457
+ // main-process pid: a bare `proc.kill()` left helper processes
458
+ // holding the profile open, which is what made rmSync lose the
459
+ // race below (see rmSyncWithRetry's own comment). Never `-pid` on
460
+ // win32 — there are no POSIX process groups there, and it isn't
461
+ // detached in the first place (chrome-launch.ts).
462
+ if (process.platform === 'win32')
463
+ proc.kill('SIGKILL');
464
+ else
465
+ process.kill(-proc.pid, 'SIGKILL');
466
+ }
467
+ }
468
+ catch {
469
+ // already gone
470
+ }
471
+ try {
472
+ deps.rmSync(userDataDir);
473
+ }
474
+ catch {
475
+ // best-effort — never let cleanup itself throw out of a signal handler
476
+ }
477
+ };
478
+ const unregisterSigint = deps.onSigint(cleanup);
479
+ let launched;
480
+ // See CaptureStage's own doc comment. Read only in the catch block below
481
+ // to decide launch_failed vs capture_failed — never assigned 'upload'.
482
+ let stage = 'launch';
483
+ try {
484
+ launched = await deps.launch(chromePath, targetUrl, userDataDir);
485
+ proc = launched.proc;
486
+ let pipeClosed = false;
487
+ const offClose = launched.cdp.onClose(() => {
488
+ pipeClosed = true;
489
+ });
490
+ stage = 'handshake';
491
+ const targetId = await deps.findPageTarget(launched.cdp);
492
+ const { closedEarly } = await deps.waitForDone(launched.cdp, targetId, cleanup);
493
+ if (pipeClosed) {
494
+ // A corrupt EVENT frame during the wait (e.g. a garbled
495
+ // targetDestroyed) tears the pipe down the same way a real Chrome
496
+ // exit does — `protocolError` distinguishes the two so this doesn't
497
+ // default to "Chrome exited" for a cause that wasn't that.
498
+ if (launched.cdp.protocolError) {
499
+ return {
500
+ ok: false,
501
+ reason: 'cdp_protocol_error',
502
+ message: cdpProtocolErrorMessage(launched.cdp.protocolError),
503
+ };
504
+ }
505
+ return {
506
+ ok: false,
507
+ reason: 'process_exited_before_capture',
508
+ message: PROCESS_EXITED_MESSAGE,
509
+ };
510
+ }
511
+ stage = 'capture';
512
+ const raw = await deps.readCapture(launched.cdp, targetUrl, closedEarly);
513
+ offClose();
514
+ const cookieResult = mapCookies(raw.cookies, targetUrl);
515
+ const originResult = mapOrigins(raw.origins, targetUrl);
516
+ // The exact host cookies/origins were scoped against above — not a
517
+ // peeled "registrable domain" (trawl_cli#183 review finding 2; see
518
+ // storage-state.ts's module doc comment for why peeling was the bug).
519
+ const targetDomain = getTargetHost(targetUrl);
520
+ if (cookieResult.cookies.length === 0) {
521
+ return {
522
+ ok: false,
523
+ reason: 'no_cookies_in_scope',
524
+ message: `0 cookies in scope for ${targetDomain} — nothing uploaded. Login had not completed on that domain when the capture ran.`,
525
+ };
526
+ }
527
+ return {
528
+ ok: true,
529
+ storageState: { cookies: cookieResult.cookies, origins: originResult.origins },
530
+ targetDomain,
531
+ counts: {
532
+ cookiesCaptured: cookieResult.cookies.length,
533
+ cookiesDroppedOutOfScope: cookieResult.droppedOutOfScope,
534
+ cookiesDroppedInvalid: cookieResult.droppedInvalid,
535
+ originsCaptured: originResult.origins.length,
536
+ originsDroppedOutOfScope: originResult.droppedOutOfScope,
537
+ originsUnreadable: raw.originsUnreadable,
538
+ closedEarly,
539
+ },
540
+ };
541
+ }
542
+ catch (e) {
543
+ // A CDP call (readCapture, most likely) can reject mid-flight because
544
+ // Chrome died, or a corrupt frame tore the pipe down, AFTER waitForDone
545
+ // already resolved normally — the `pipeClosed` check above only covers
546
+ // the window up to that point. `protocolError` is checked FIRST (it
547
+ // implies `isClosed` too) so a corrupt-frame failure reports its own
548
+ // factual reason instead of the generic "Chrome exited"; `isClosed`
549
+ // alone still distinguishes a real exit from every other unexpected
550
+ // failure here (a malformed response, a genuinely broken launch).
551
+ if (launched?.cdp.protocolError) {
552
+ return {
553
+ ok: false,
554
+ reason: 'cdp_protocol_error',
555
+ message: cdpProtocolErrorMessage(launched.cdp.protocolError),
556
+ };
557
+ }
558
+ if (launched?.cdp.isClosed) {
559
+ return {
560
+ ok: false,
561
+ reason: 'process_exited_before_capture',
562
+ message: PROCESS_EXITED_MESSAGE,
563
+ };
564
+ }
565
+ // trawl_cli#183 R4: everything above already special-cases a dead pipe
566
+ // (real exit or a corrupt frame). What's left here — the pipe still
567
+ // open, no protocol error — used to fall into one undifferentiated
568
+ // `launch_failed` regardless of WHEN it happened. That mislabels a
569
+ // `readCapture` failure that occurs after launch + handshake already
570
+ // succeeded (Chrome reachable over CDP, a page already open, the human
571
+ // already done logging in): every LEGITIMATE per-page CDP rejection at
572
+ // that point is already counted as `originsUnreadable` by
573
+ // `readCaptureDefault`'s own provenance boundary (see
574
+ // `readPageOverCdp`'s doc comment) — so anything that still reaches
575
+ // here from the `capture` stage is a genuine bug in THIS CLI (or an
576
+ // unanticipated non-page-scoped failure, e.g. `Storage.getCookies`
577
+ // itself), never a "Chrome could not be launched/driven" problem.
578
+ // `launch`/`handshake` stay `launch_failed`, message unchanged.
579
+ if (stage === 'capture') {
580
+ // Never `(e as Error).message` here: the whole point of the R3/R4
581
+ // history (trawl_cli#183 post-cap review finding B, then R4) is that
582
+ // a bug in THIS FILE's own code can be TRIGGERED by page-influenced
583
+ // data (a tampered global, a crafted return value) — nothing
584
+ // guarantees the resulting exception's own message is free of it, so
585
+ // the stage name is the only safe, fixed vocabulary to report here.
586
+ // No counts either: `readCapture` rejected before returning any —
587
+ // there is no partial `raw` to report counts from.
588
+ return {
589
+ ok: false,
590
+ reason: 'capture_failed',
591
+ message: 'The capture failed during the capture stage, after Chrome was already reachable over CDP and a page was already open — this is a CLI-side failure, not a page or environment problem. Nothing was uploaded.',
592
+ };
593
+ }
594
+ return { ok: false, reason: 'launch_failed', message: `Chrome could not be driven over CDP: ${e.message}` };
595
+ }
596
+ finally {
597
+ unregisterSigint();
598
+ cleanup();
599
+ }
600
+ }