@ultimat3/scraping 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +194 -0
- package/package.json +38 -0
- package/src/actionability.ts +106 -0
- package/src/artifacts.ts +69 -0
- package/src/auth.ts +200 -0
- package/src/cdp-fake.ts +150 -0
- package/src/cdp-port.ts +75 -0
- package/src/cdp-snapshot.ts +63 -0
- package/src/cdp-target.ts +320 -0
- package/src/clock.ts +97 -0
- package/src/cookie-scope.ts +97 -0
- package/src/driver-cdp.ts +184 -0
- package/src/driver-fake.ts +119 -0
- package/src/driver-fixture.ts +65 -0
- package/src/driver.ts +88 -0
- package/src/error-throws.ts +258 -0
- package/src/errors.ts +180 -0
- package/src/events.ts +74 -0
- package/src/expect.ts +133 -0
- package/src/failures.ts +47 -0
- package/src/hosts.ts +56 -0
- package/src/html-query.ts +119 -0
- package/src/html-requests.ts +45 -0
- package/src/html-target.ts +229 -0
- package/src/http-recorded.ts +85 -0
- package/src/http.ts +158 -0
- package/src/index.ts +178 -0
- package/src/intercept.ts +41 -0
- package/src/offline-session.ts +67 -0
- package/src/page-over-target.ts +215 -0
- package/src/page.ts +103 -0
- package/src/rate.ts +23 -0
- package/src/recording.ts +68 -0
- package/src/recover.ts +51 -0
- package/src/rings.ts +73 -0
- package/src/robots.ts +144 -0
- package/src/scrape-run.ts +226 -0
- package/src/scrape.ts +151 -0
- package/src/secrets.ts +91 -0
- package/src/session-state.ts +181 -0
- package/src/target.ts +118 -0
- package/src/watchdog.ts +100 -0
package/src/clock.ts
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
// The ONE file in this package allowed to make time pass. Every wait a scraper does — the
|
|
2
|
+
// actionability poll, the wedge watchdog, the rate limiter, the graceful-quit ceiling — goes
|
|
3
|
+
// through `ScrapeClock`, so a suite that exercises a 30-second timeout finishes in microseconds.
|
|
4
|
+
//
|
|
5
|
+
// `clock-discipline.test.ts` fails the build if a second file reaches for a timer directly.
|
|
6
|
+
|
|
7
|
+
import type { Clock } from '@ultimat3/core';
|
|
8
|
+
|
|
9
|
+
export interface ScrapeClock extends Clock {
|
|
10
|
+
/**
|
|
11
|
+
* Resolve after `ms` of this clock's time, or reject with the signal's reason the moment it
|
|
12
|
+
* aborts. Every poll loop here composes a wait with a cancellation, so a clock whose sleep
|
|
13
|
+
* ignored the signal would leave a killed run still counting.
|
|
14
|
+
*/
|
|
15
|
+
sleep(ms: number, signal?: AbortSignal): Promise<void>;
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
export const systemScrapeClock: ScrapeClock = Object.freeze({
|
|
19
|
+
now(): Date {
|
|
20
|
+
return new Date();
|
|
21
|
+
},
|
|
22
|
+
monotonic(): number {
|
|
23
|
+
return performance.now();
|
|
24
|
+
},
|
|
25
|
+
async sleep(ms: number, signal?: AbortSignal): Promise<void> {
|
|
26
|
+
if (signal === undefined) {
|
|
27
|
+
await Bun.sleep(ms);
|
|
28
|
+
return;
|
|
29
|
+
}
|
|
30
|
+
throwIfAborted(signal);
|
|
31
|
+
// A timer raced against the signal, never `Bun.sleep(ms).then(check)`: the watchdog aborts a
|
|
32
|
+
// wedged run precisely so nothing waits out the rest of a five-minute budget, and a sleep
|
|
33
|
+
// that only notices the abort when it expires would wait out every one of them.
|
|
34
|
+
await new Promise<void>((resolve, reject) => {
|
|
35
|
+
// The reason is rejected VERBATIM, never wrapped: whoever aborted put a `ScrapeError` with
|
|
36
|
+
// a code and a fix there, and re-wrapping it would replace an instruction with `Error`.
|
|
37
|
+
const onAbort = (): void => {
|
|
38
|
+
clearTimeout(timer);
|
|
39
|
+
reject(signal.reason);
|
|
40
|
+
};
|
|
41
|
+
const timer = setTimeout(() => {
|
|
42
|
+
signal.removeEventListener('abort', onAbort);
|
|
43
|
+
resolve();
|
|
44
|
+
}, ms);
|
|
45
|
+
signal.addEventListener('abort', onAbort, { once: true });
|
|
46
|
+
});
|
|
47
|
+
},
|
|
48
|
+
});
|
|
49
|
+
|
|
50
|
+
/** The abort reason, unwrapped — a caller's `AbortSignal.reason` is whatever they put there. */
|
|
51
|
+
export function throwIfAborted(signal: AbortSignal): void {
|
|
52
|
+
if (signal.aborted) throw signal.reason;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* A clock in which sleeping IS advancing: `await clock.sleep(30_000)` returns on the next
|
|
57
|
+
* microtask with thirty seconds elapsed. This is what makes a poll-until-deadline test instant
|
|
58
|
+
* without the test knowing how many polls the loop makes — a count a test that asserted it would
|
|
59
|
+
* pin to the implementation rather than to the behaviour.
|
|
60
|
+
*/
|
|
61
|
+
export interface TestScrapeClock extends ScrapeClock {
|
|
62
|
+
advance(ms: number): void;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
export function testClock(at: Date | number = 0): TestScrapeClock {
|
|
66
|
+
let epochMs = at instanceof Date ? at.getTime() : at;
|
|
67
|
+
let mono = 0;
|
|
68
|
+
const advance = (ms: number): void => {
|
|
69
|
+
epochMs += ms;
|
|
70
|
+
mono += ms;
|
|
71
|
+
};
|
|
72
|
+
return {
|
|
73
|
+
now: () => new Date(epochMs),
|
|
74
|
+
monotonic: () => mono,
|
|
75
|
+
advance,
|
|
76
|
+
sleep: async (ms: number, signal?: AbortSignal) => {
|
|
77
|
+
advance(ms);
|
|
78
|
+
// Yield the microtask queue anyway: a loop that never awaits anything real starves whatever
|
|
79
|
+
// the test armed to cancel it, and the wedge tests arm exactly that.
|
|
80
|
+
await Promise.resolve();
|
|
81
|
+
if (signal !== undefined) throwIfAborted(signal);
|
|
82
|
+
},
|
|
83
|
+
};
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/** Elapsed-time budget, read from one clock. Built once per waiting call, never shared. */
|
|
87
|
+
export interface Deadline {
|
|
88
|
+
readonly totalMs: number;
|
|
89
|
+
remainingMs(): number;
|
|
90
|
+
expired(): boolean;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
export function deadline(clock: Clock, totalMs: number): Deadline {
|
|
94
|
+
const startedAt = clock.monotonic();
|
|
95
|
+
const remainingMs = (): number => Math.max(0, totalMs - (clock.monotonic() - startedAt));
|
|
96
|
+
return { totalMs, remainingMs, expired: () => remainingMs() <= 0 };
|
|
97
|
+
}
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
// Which cookies a URL may see — RFC 6265 §5.1.3 (domain-match) and §5.1.4 (path-match), as one
|
|
2
|
+
// decision every transport asks.
|
|
3
|
+
//
|
|
4
|
+
// It is a SECURITY rule and not a formatting one: a session snapshot's jar is `browser.cookies()`,
|
|
5
|
+
// i.e. every domain the session ever touched — an SSO hop's included — and the HTTP leg picks from
|
|
6
|
+
// it by hand rather than by a browser's own jar. A suffix test with no dot boundary sends a
|
|
7
|
+
// `bank.test` session cookie to `evilbank.test`; one with no host-only rule sends it to
|
|
8
|
+
// `sub.bank.test`. Both are the same one-line mistake, in opposite directions.
|
|
9
|
+
|
|
10
|
+
import type { ScrapeCookie } from './target';
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* The dot IS the rule, in both directions.
|
|
14
|
+
*
|
|
15
|
+
* A stored `.bank.test` is DOMAIN-scoped — a browser records the leading dot for a cookie set with
|
|
16
|
+
* a `Domain=` attribute, and that is what CDP's `Network.getAllCookies` hands back — so it reaches
|
|
17
|
+
* `bank.test` and any subdomain of it. A stored `bank.test` is HOST-ONLY and reaches exactly that
|
|
18
|
+
* host. `ScrapeCookie` carries no `hostOnly` flag because the CDP cookie shape has none to carry
|
|
19
|
+
* (`cdp-port.ts`), so the leading dot is the only signal there is, and it is enough.
|
|
20
|
+
*/
|
|
21
|
+
export function cookieDomainMatches(host: string, domain: string): boolean {
|
|
22
|
+
const requested = host.trim().toLowerCase();
|
|
23
|
+
const stored = domain.trim().toLowerCase();
|
|
24
|
+
const scoped = stored.startsWith('.');
|
|
25
|
+
const bare = scoped ? stored.slice(1) : stored;
|
|
26
|
+
if (bare === '' || requested === '') return false;
|
|
27
|
+
if (requested === bare) return true;
|
|
28
|
+
return scoped && requested.endsWith(`.${bare}`);
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* RFC 6265 §5.1.4. `/admin` covers `/admin` and `/admin/users`, and never `/administrators` —
|
|
33
|
+
* the boundary is a `/`, exactly as it is for a domain. An empty stored path is `/`, which is what
|
|
34
|
+
* a jar entry written by hand usually means.
|
|
35
|
+
*/
|
|
36
|
+
export function cookiePathMatches(requestPath: string, cookiePath: string): boolean {
|
|
37
|
+
const wanted = requestPath === '' ? '/' : requestPath;
|
|
38
|
+
const stored = cookiePath === '' ? '/' : cookiePath;
|
|
39
|
+
if (wanted === stored) return true;
|
|
40
|
+
if (!wanted.startsWith(stored)) return false;
|
|
41
|
+
return stored.endsWith('/') || wanted.charAt(stored.length) === '/';
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* A `secure` cookie over plaintext is the same leak one hop further down: `http:` on a hostile
|
|
46
|
+
* network is readable. `localhost` is the one exception every browser makes, and a fixture host
|
|
47
|
+
* that is not is simply refused the cookie rather than silently downgraded.
|
|
48
|
+
*/
|
|
49
|
+
const trustworthy = (url: URL): boolean =>
|
|
50
|
+
url.protocol === 'https:' ||
|
|
51
|
+
url.hostname === 'localhost' ||
|
|
52
|
+
url.hostname === '127.0.0.1' ||
|
|
53
|
+
url.hostname === '[::1]';
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Fails CLOSED, like `hostDecision()`: a URL that will not parse gets no cookies at all. Expiry is
|
|
57
|
+
* deliberately NOT filtered here — `ScrapeCookie.expires` carries no unit (CDP answers seconds,
|
|
58
|
+
* a hand-written jar tends to hold milliseconds) and dropping a live session cookie over a guess
|
|
59
|
+
* is worse than sending one the site will refuse itself.
|
|
60
|
+
*/
|
|
61
|
+
export function cookiesForUrl(
|
|
62
|
+
cookies: readonly ScrapeCookie[],
|
|
63
|
+
url: string,
|
|
64
|
+
): readonly ScrapeCookie[] {
|
|
65
|
+
let parsed: URL;
|
|
66
|
+
try {
|
|
67
|
+
parsed = new URL(url);
|
|
68
|
+
} catch {
|
|
69
|
+
return [];
|
|
70
|
+
}
|
|
71
|
+
const secureOk = trustworthy(parsed);
|
|
72
|
+
return cookies.filter(
|
|
73
|
+
(cookie) =>
|
|
74
|
+
cookieDomainMatches(parsed.hostname, cookie.domain) &&
|
|
75
|
+
cookiePathMatches(parsed.pathname, cookie.path) &&
|
|
76
|
+
(!cookie.secure || secureOk),
|
|
77
|
+
);
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/** An absent path is `/` here too — §5.1.4's rule, so it cannot outrank a real path on length. */
|
|
81
|
+
const pathLength = (path: string): number => (path === '' ? 1 : path.length);
|
|
82
|
+
|
|
83
|
+
/**
|
|
84
|
+
* The `cookie:` header this URL earns, or `undefined` when the jar has nothing for it.
|
|
85
|
+
*
|
|
86
|
+
* ORDERED, RFC 6265 §5.4: longer paths first. Two cookies may share a name — a `sid` at `/` and a
|
|
87
|
+
* `sid` at `/admin` — and a server reading the first occurrence has to see the specific one; jar
|
|
88
|
+
* order is an accident of how the browser filled it. The sort is stable, so a tie keeps jar order,
|
|
89
|
+
* which is the nearest thing this jar has to §5.4's creation-time tiebreak (`ScrapeCookie` carries
|
|
90
|
+
* no creation time, and CDP's cookie shape has none to carry).
|
|
91
|
+
*/
|
|
92
|
+
export function cookieHeaderFor(cookies: readonly ScrapeCookie[], url: string): string | undefined {
|
|
93
|
+
const jar = [...cookiesForUrl(cookies, url)].sort(
|
|
94
|
+
(left, right) => pathLength(right.path) - pathLength(left.path),
|
|
95
|
+
);
|
|
96
|
+
return jar.length === 0 ? undefined : jar.map((c) => `${c.name}=${c.value}`).join('; ');
|
|
97
|
+
}
|
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
// The real browser driver: `localBrowser()` starts one in this container, `remoteBrowser()`
|
|
2
|
+
// ATTACHES to one somebody else started over a CDP URL.
|
|
3
|
+
//
|
|
4
|
+
// Attach is the PRIMARY production path, not an afterthought. Real deployments create a hardened
|
|
5
|
+
// browser elsewhere — a provider, a sidecar, a stealth build — and the app connects to it. Which
|
|
6
|
+
// is why `close()` here stops BOTH halves: the local connection and the remote session. A close
|
|
7
|
+
// that only disconnects leaves a browser somebody is billing for running until its provider times
|
|
8
|
+
// it out, and nobody attributes that bill to the run that caused it.
|
|
9
|
+
//
|
|
10
|
+
// The library is passed IN (`launcher`), never imported — see `cdp-port.ts` for why that seam is
|
|
11
|
+
// what keeps a puppeteer type out of the vocabulary and this package free of a dependency.
|
|
12
|
+
|
|
13
|
+
import type { CdpBrowserLike, CdpLauncherLike } from './cdp-port';
|
|
14
|
+
import { CDP_DRIVER, cdpTarget } from './cdp-target';
|
|
15
|
+
import type { ScrapeDriver, ScrapeSession, SessionInit } from './driver';
|
|
16
|
+
import { browserUnreachable, cdpAttachFailed, remoteRequired } from './error-throws';
|
|
17
|
+
import { isScrapeError } from './errors';
|
|
18
|
+
import { httpOverFetch } from './http';
|
|
19
|
+
import { pageOverTarget } from './page-over-target';
|
|
20
|
+
import type { ScrapeTarget } from './target';
|
|
21
|
+
import { createWedgeGuard } from './watchdog';
|
|
22
|
+
|
|
23
|
+
export { CDP_DRIVER } from './cdp-target';
|
|
24
|
+
|
|
25
|
+
export interface BrowserOptions {
|
|
26
|
+
/** `puppeteer` itself, or anything with the same two methods. */
|
|
27
|
+
readonly launcher: CdpLauncherLike;
|
|
28
|
+
readonly proxy?: string | undefined;
|
|
29
|
+
/** Extra launch/connect arguments, passed through untouched. */
|
|
30
|
+
readonly options?: Record<string, unknown> | undefined;
|
|
31
|
+
/** How long a graceful `close()` may take before the process is killed. */
|
|
32
|
+
readonly graceMs?: number | undefined;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
export interface LocalBrowserOptions extends BrowserOptions {
|
|
36
|
+
readonly executablePath?: string | undefined;
|
|
37
|
+
readonly headless?: boolean | undefined;
|
|
38
|
+
/**
|
|
39
|
+
* The user-data directory. Two runs sharing one is `X_SCRAPE_PROFILE_LOCKED`; a run that must
|
|
40
|
+
* arrive as a NEW identity gets its own, which is what burning a session means on disk.
|
|
41
|
+
*/
|
|
42
|
+
readonly profileDir?: string | undefined;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
export interface RemoteBrowserOptions extends BrowserOptions {
|
|
46
|
+
/** The `webSocketDebuggerUrl` from `/json/version`, or a provider's connect URL. */
|
|
47
|
+
readonly cdpUrl: string;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* The page, its interception and its restored session — or a closed browser and the failure.
|
|
52
|
+
* A throw from here is classified before it leaves: `newPage()` and `setRequestInterception()` are
|
|
53
|
+
* outside `cdpTarget`'s own `guard()`, so a bare library `Error` would otherwise reach the job's
|
|
54
|
+
* retry classifier with no code at all.
|
|
55
|
+
*/
|
|
56
|
+
async function opened(browser: CdpBrowserLike, init: SessionInit): Promise<ScrapeTarget> {
|
|
57
|
+
try {
|
|
58
|
+
const page = await browser.newPage();
|
|
59
|
+
const target = await cdpTarget({ page, browser, rules: init.rules, clock: init.clock });
|
|
60
|
+
if (init.restore !== undefined) await target.restore(init.restore);
|
|
61
|
+
return target;
|
|
62
|
+
} catch (thrown) {
|
|
63
|
+
// Best effort, and it may never replace the failure that caused it: a close that also throws
|
|
64
|
+
// would hide the tab limit or the refused interception the reader actually needs.
|
|
65
|
+
await browser.close().catch(() => undefined);
|
|
66
|
+
throw isScrapeError(thrown) ? thrown : browserUnreachable(CDP_DRIVER, thrown);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
async function sessionOver(
|
|
71
|
+
browser: CdpBrowserLike,
|
|
72
|
+
init: SessionInit,
|
|
73
|
+
options: BrowserOptions,
|
|
74
|
+
): Promise<ScrapeSession> {
|
|
75
|
+
// Acquire, then roll back on ANY throw — the shape `releaseBoot` uses in `packages/cli/src/
|
|
76
|
+
// serve.ts`. Between the launch and the `WedgeGuard` below, nothing else holds this browser:
|
|
77
|
+
// `runScrape`'s `finally { session.close() }` never runs for a session `open()` did not return,
|
|
78
|
+
// so a tab limit, a refused interception or a restore that threw left a real Chrome process —
|
|
79
|
+
// or a remote session somebody is billing for — running per attempt, unattributed.
|
|
80
|
+
const target = await opened(browser, init);
|
|
81
|
+
const guard = createWedgeGuard({
|
|
82
|
+
clock: init.clock,
|
|
83
|
+
what: `scrape "${init.name}"`,
|
|
84
|
+
graceMs: init.watchdog?.graceMs ?? options.graceMs,
|
|
85
|
+
idleMs: init.watchdog?.idleMs,
|
|
86
|
+
// `close()` and NOT `disconnect()`, on BOTH drivers. Disconnecting ends the local half and
|
|
87
|
+
// leaves the remote browser running until its provider times it out — a bill nobody
|
|
88
|
+
// attributes to the run that caused it. An app that genuinely wants the remote session to
|
|
89
|
+
// survive keeps its own handle and never hands it to a driver.
|
|
90
|
+
quit: () => browser.close(),
|
|
91
|
+
kill: () => {
|
|
92
|
+
// The OS process, when there is one to reach. Killing it is what makes the socket a wedged
|
|
93
|
+
// await is blocked on close, which is what turns an infinite wait into a catchable error.
|
|
94
|
+
const child = browser.process?.();
|
|
95
|
+
child?.kill(9);
|
|
96
|
+
},
|
|
97
|
+
});
|
|
98
|
+
const onActivity = (): void => {
|
|
99
|
+
guard.touch();
|
|
100
|
+
init.onActivity?.();
|
|
101
|
+
};
|
|
102
|
+
return {
|
|
103
|
+
driver: CDP_DRIVER,
|
|
104
|
+
page: pageOverTarget(target, {
|
|
105
|
+
clock: init.clock,
|
|
106
|
+
allowHosts: init.rules.allowHosts,
|
|
107
|
+
defaultTimeoutMs: init.timeoutMs,
|
|
108
|
+
secrets: init.secrets,
|
|
109
|
+
robots: init.robots,
|
|
110
|
+
signal: init.signal,
|
|
111
|
+
onActivity,
|
|
112
|
+
pace: init.pace,
|
|
113
|
+
}),
|
|
114
|
+
http: httpOverFetch({
|
|
115
|
+
rules: init.rules,
|
|
116
|
+
clock: init.clock,
|
|
117
|
+
timeoutMs: init.timeoutMs,
|
|
118
|
+
network: target.network,
|
|
119
|
+
// Straight through to the live browser: the HTTP leg must see the cookies a login two calls
|
|
120
|
+
// ago produced, and a snapshot taken at open time would be the logged-out one forever.
|
|
121
|
+
session: () => target.session(),
|
|
122
|
+
robots: init.robots,
|
|
123
|
+
pace: init.pace,
|
|
124
|
+
signal: init.signal,
|
|
125
|
+
onActivity,
|
|
126
|
+
proxy: options.proxy,
|
|
127
|
+
}),
|
|
128
|
+
close: () => guard.shutdown(),
|
|
129
|
+
};
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
/**
|
|
133
|
+
* A browser in this container. `executablePath` is required by every puppeteer-core build — it
|
|
134
|
+
* ships no browser — so it is passed through rather than guessed at.
|
|
135
|
+
*/
|
|
136
|
+
export function localBrowser(options: LocalBrowserOptions): ScrapeDriver {
|
|
137
|
+
return {
|
|
138
|
+
name: CDP_DRIVER,
|
|
139
|
+
async open(init: SessionInit): Promise<ScrapeSession> {
|
|
140
|
+
const launch = options.launcher.launch;
|
|
141
|
+
if (launch === undefined) {
|
|
142
|
+
throw remoteRequired('local browser: the launcher has no launch()');
|
|
143
|
+
}
|
|
144
|
+
let browser: CdpBrowserLike;
|
|
145
|
+
try {
|
|
146
|
+
browser = await launch.call(options.launcher, {
|
|
147
|
+
headless: options.headless ?? true,
|
|
148
|
+
...(options.executablePath === undefined
|
|
149
|
+
? {}
|
|
150
|
+
: { executablePath: options.executablePath }),
|
|
151
|
+
...(options.profileDir === undefined ? {} : { userDataDir: options.profileDir }),
|
|
152
|
+
...(options.proxy === undefined ? {} : { args: [`--proxy-server=${options.proxy}`] }),
|
|
153
|
+
...options.options,
|
|
154
|
+
});
|
|
155
|
+
} catch (thrown) {
|
|
156
|
+
throw browserUnreachable(CDP_DRIVER, thrown);
|
|
157
|
+
}
|
|
158
|
+
return sessionOver(browser, init, options);
|
|
159
|
+
},
|
|
160
|
+
};
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/** Attach to a browser somebody else started. The production path. */
|
|
164
|
+
export function remoteBrowser(options: RemoteBrowserOptions): ScrapeDriver {
|
|
165
|
+
return {
|
|
166
|
+
name: CDP_DRIVER,
|
|
167
|
+
async open(init: SessionInit): Promise<ScrapeSession> {
|
|
168
|
+
const connect = options.launcher.connect;
|
|
169
|
+
if (connect === undefined || options.cdpUrl === '') {
|
|
170
|
+
throw remoteRequired(CDP_DRIVER);
|
|
171
|
+
}
|
|
172
|
+
let browser: CdpBrowserLike;
|
|
173
|
+
try {
|
|
174
|
+
browser = await connect.call(options.launcher, {
|
|
175
|
+
browserWSEndpoint: options.cdpUrl,
|
|
176
|
+
...options.options,
|
|
177
|
+
});
|
|
178
|
+
} catch (thrown) {
|
|
179
|
+
throw cdpAttachFailed(options.cdpUrl, thrown);
|
|
180
|
+
}
|
|
181
|
+
return sessionOver(browser, init, options);
|
|
182
|
+
},
|
|
183
|
+
};
|
|
184
|
+
}
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
// The DEFAULT driver under `bun test`: no process, no port, no CDP, no Chrome. A scraper's tests
|
|
2
|
+
// are the point of this package, and a test suite that needs a browser installed is a test suite
|
|
3
|
+
// that runs on one machine.
|
|
4
|
+
//
|
|
5
|
+
// It covers BOTH legs. `fakeBrowser({ pages, http })` replays the browser walk and the JSON
|
|
6
|
+
// endpoints behind it from one declaration, so a hybrid scrape is tested the way it runs.
|
|
7
|
+
|
|
8
|
+
import type { ScrapeClock } from './clock';
|
|
9
|
+
import { systemScrapeClock } from './clock';
|
|
10
|
+
import type { ScrapeDriver, ScrapeSession, SessionInit } from './driver';
|
|
11
|
+
import { htmlTarget } from './html-target';
|
|
12
|
+
import { httpRecordingsOf } from './http-recorded';
|
|
13
|
+
import { openOfflineSession } from './offline-session';
|
|
14
|
+
import type { ScrapePage } from './page';
|
|
15
|
+
import type { PageContext } from './page-over-target';
|
|
16
|
+
import { pageOverTarget } from './page-over-target';
|
|
17
|
+
import type { HttpRecording, PageRecording } from './recording';
|
|
18
|
+
import type { SessionSnapshot } from './session-state';
|
|
19
|
+
import type { ScrapeCookie } from './target';
|
|
20
|
+
|
|
21
|
+
export const FAKE_DRIVER = 'fake';
|
|
22
|
+
|
|
23
|
+
/** `{ 'https://example.com/': '<html>…' }`, or full recordings when a page needs frames. */
|
|
24
|
+
export type FakePages = Readonly<Record<string, string>> | readonly PageRecording[];
|
|
25
|
+
|
|
26
|
+
const normalise = (url: string): string => {
|
|
27
|
+
try {
|
|
28
|
+
const parsed = new URL(url);
|
|
29
|
+
// `https://example.com` and `https://example.com/` are one page to every browser, and two
|
|
30
|
+
// keys to a `Map` — which is a `X_SCRAPE_FIXTURE_MISSING` for a page the author did record.
|
|
31
|
+
if (parsed.pathname === '') parsed.pathname = '/';
|
|
32
|
+
return parsed.toString();
|
|
33
|
+
} catch {
|
|
34
|
+
return url;
|
|
35
|
+
}
|
|
36
|
+
};
|
|
37
|
+
|
|
38
|
+
export function recordingsOf(pages: FakePages): readonly PageRecording[] {
|
|
39
|
+
return Array.isArray(pages)
|
|
40
|
+
? (pages as readonly PageRecording[])
|
|
41
|
+
: Object.entries(pages as Readonly<Record<string, string>>).map(([url, html]) => ({
|
|
42
|
+
url,
|
|
43
|
+
html,
|
|
44
|
+
}));
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
export interface FakeBrowserOptions {
|
|
48
|
+
readonly cookies?: readonly ScrapeCookie[] | undefined;
|
|
49
|
+
/** What `page.session()` answers — the browser-to-HTTP handoff, as data a test can assert on. */
|
|
50
|
+
readonly session?: SessionSnapshot | undefined;
|
|
51
|
+
/** The JSON endpoints the hybrid leg calls. An unrecorded one throws, like an unrecorded page. */
|
|
52
|
+
readonly http?: readonly HttpRecording[] | undefined;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* An in-memory site. Every request it cannot answer — page or HTTP — THROWS
|
|
57
|
+
* `X_SCRAPE_FIXTURE_MISSING`, never a pass-through to the network, which is the one behaviour
|
|
58
|
+
* that would make an offline suite secretly live.
|
|
59
|
+
*/
|
|
60
|
+
export function fakeBrowser(pages: FakePages, options: FakeBrowserOptions = {}): ScrapeDriver {
|
|
61
|
+
const byUrl = new Map(recordingsOf(pages).map((page) => [normalise(page.url), page]));
|
|
62
|
+
const byRequest = httpRecordingsOf(options.http ?? []);
|
|
63
|
+
return {
|
|
64
|
+
name: FAKE_DRIVER,
|
|
65
|
+
open: (session: SessionInit): Promise<ScrapeSession> =>
|
|
66
|
+
openOfflineSession({
|
|
67
|
+
driver: FAKE_DRIVER,
|
|
68
|
+
source: 'fakeBrowser()',
|
|
69
|
+
lookup: (url) => Promise.resolve(byUrl.get(normalise(url))),
|
|
70
|
+
http: (method, url) => Promise.resolve(byRequest.get(`${method} ${url}`)),
|
|
71
|
+
session,
|
|
72
|
+
cookies: options.cookies,
|
|
73
|
+
snapshot: options.session,
|
|
74
|
+
}),
|
|
75
|
+
};
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
export interface FakePageOptions {
|
|
79
|
+
readonly url?: string | undefined;
|
|
80
|
+
readonly clock?: ScrapeClock | undefined;
|
|
81
|
+
readonly allowHosts?: readonly string[] | undefined;
|
|
82
|
+
readonly timeoutMs?: number | undefined;
|
|
83
|
+
readonly context?: Partial<PageContext> | undefined;
|
|
84
|
+
/** Extra pages this one can navigate to — `data-goto` and `<a href>` both land here. */
|
|
85
|
+
readonly pages?: FakePages | undefined;
|
|
86
|
+
readonly session?: SessionSnapshot | undefined;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
export const FAKE_PAGE_URL = 'https://fake.test/';
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* One page of markup as a `ScrapePage`, for a test that wants to assert on the vocabulary and
|
|
93
|
+
* nothing else. No session and no driver: `page.html()`, `page.click()` and `page.values()` are
|
|
94
|
+
* the whole surface under test.
|
|
95
|
+
*/
|
|
96
|
+
export function fakePage(dom: string, options: FakePageOptions = {}): ScrapePage {
|
|
97
|
+
const url = options.url ?? FAKE_PAGE_URL;
|
|
98
|
+
const start: PageRecording = { url, html: dom };
|
|
99
|
+
const clock = options.clock ?? systemScrapeClock;
|
|
100
|
+
const allowHosts = options.allowHosts ?? ['*'];
|
|
101
|
+
const byUrl = new Map(
|
|
102
|
+
[start, ...recordingsOf(options.pages ?? [])].map((page) => [normalise(page.url), page]),
|
|
103
|
+
);
|
|
104
|
+
const target = htmlTarget({
|
|
105
|
+
driver: FAKE_DRIVER,
|
|
106
|
+
lookup: (next) => Promise.resolve(byUrl.get(normalise(next))),
|
|
107
|
+
rules: { allowHosts },
|
|
108
|
+
clock,
|
|
109
|
+
source: 'fakePage()',
|
|
110
|
+
start,
|
|
111
|
+
session: options.session,
|
|
112
|
+
});
|
|
113
|
+
return pageOverTarget(target, {
|
|
114
|
+
clock,
|
|
115
|
+
allowHosts,
|
|
116
|
+
defaultTimeoutMs: options.timeoutMs ?? 1_000,
|
|
117
|
+
...options.context,
|
|
118
|
+
});
|
|
119
|
+
}
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
// Recorded pages AND recorded HTTP on disk, replayed from ONE directory. The difference from
|
|
2
|
+
// `fakeBrowser()` is only where the recordings live — and the difference from a live driver is
|
|
3
|
+
// that this one CANNOT reach the network: a request with no file under `dir` is
|
|
4
|
+
// `X_SCRAPE_FIXTURE_MISSING`, never a fetch.
|
|
5
|
+
//
|
|
6
|
+
// One directory for both legs is the point. A hybrid scrape — browser login, session handoff,
|
|
7
|
+
// HTTP bulk fetch — replays end to end, so the part of a scraper that is hardest to get right is
|
|
8
|
+
// not the part that has no test.
|
|
9
|
+
//
|
|
10
|
+
// Recordings age. A fixture recorded eighteen months ago proves that a scraper still parses a
|
|
11
|
+
// page that no longer exists, which is worse than no test at all — hence `maxAge`.
|
|
12
|
+
|
|
13
|
+
import type { ScrapeDriver, ScrapeSession, SessionInit } from './driver';
|
|
14
|
+
import { httpRecordingFilename } from './http-recorded';
|
|
15
|
+
import { openOfflineSession } from './offline-session';
|
|
16
|
+
import type { HttpRecording, PageRecording } from './recording';
|
|
17
|
+
import { parseHttpRecording, parseRecording } from './recording';
|
|
18
|
+
|
|
19
|
+
export const FIXTURE_DRIVER = 'fixture';
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* `https://example.com/a/b?q=1` -> `example.com-a-b-q-1.json`. Deterministic and readable, so the
|
|
23
|
+
* `X_SCRAPE_FIXTURE_MISSING` fix can name the exact file to create and a reviewer can tell which
|
|
24
|
+
* page a recording is without opening it.
|
|
25
|
+
*/
|
|
26
|
+
export function recordingFilename(url: string): string {
|
|
27
|
+
const withoutScheme = url.replace(/^[a-z]+:\/\//i, '');
|
|
28
|
+
const slug = withoutScheme
|
|
29
|
+
.toLowerCase()
|
|
30
|
+
.replaceAll(/[^a-z0-9]+/g, '-')
|
|
31
|
+
.replace(/^-+|-+$/g, '');
|
|
32
|
+
return `${slug === '' ? 'index' : slug}.json`;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
export interface FixtureBrowserOptions {
|
|
36
|
+
/** Milliseconds. A recording older than this refuses the run rather than passing on old HTML. */
|
|
37
|
+
readonly maxAge?: number | undefined;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
export function fixtureBrowser(dir: string, options: FixtureBrowserOptions = {}): ScrapeDriver {
|
|
41
|
+
const lookup = async (url: string): Promise<PageRecording | undefined> => {
|
|
42
|
+
const file = Bun.file(`${dir}/${recordingFilename(url)}`);
|
|
43
|
+
if (!(await file.exists())) return undefined;
|
|
44
|
+
// Parsed, never cast: a recording is an edited file on somebody's disk, so it is `unknown`
|
|
45
|
+
// until a schema says otherwise.
|
|
46
|
+
return parseRecording(await file.json());
|
|
47
|
+
};
|
|
48
|
+
const http = async (method: string, url: string): Promise<HttpRecording | undefined> => {
|
|
49
|
+
const file = Bun.file(`${dir}/${httpRecordingFilename(method, url)}`);
|
|
50
|
+
if (!(await file.exists())) return undefined;
|
|
51
|
+
return parseHttpRecording(await file.json());
|
|
52
|
+
};
|
|
53
|
+
return {
|
|
54
|
+
name: FIXTURE_DRIVER,
|
|
55
|
+
open: (session: SessionInit): Promise<ScrapeSession> =>
|
|
56
|
+
openOfflineSession({
|
|
57
|
+
driver: FIXTURE_DRIVER,
|
|
58
|
+
source: dir,
|
|
59
|
+
lookup,
|
|
60
|
+
http,
|
|
61
|
+
session,
|
|
62
|
+
maxAgeMs: options.maxAge,
|
|
63
|
+
}),
|
|
64
|
+
};
|
|
65
|
+
}
|
package/src/driver.ts
ADDED
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
// The browser contract. Every driver implements exactly this, so a scrape's body never names one
|
|
2
|
+
// — the same shape `packages/jobs/src/driver.ts` gives the queue, for the same reason: swapping
|
|
3
|
+
// the browser is `setScrapeDriver(other)` and ZERO run-body change.
|
|
4
|
+
//
|
|
5
|
+
// Two methods. `open` hands back a session; `close` ends it. Everything else a scraper does goes
|
|
6
|
+
// through `ScrapePage`, which is driver-blind by construction.
|
|
7
|
+
|
|
8
|
+
import type { ScrapeClock } from './clock';
|
|
9
|
+
import type { ScrapeHttp } from './http';
|
|
10
|
+
import type { InterceptRules } from './intercept';
|
|
11
|
+
import type { ScrapePage } from './page';
|
|
12
|
+
import type { RobotsGate } from './robots';
|
|
13
|
+
import type { ScrapeSecrets } from './secrets';
|
|
14
|
+
import type { SessionSnapshot } from './session-state';
|
|
15
|
+
|
|
16
|
+
export interface SessionInit {
|
|
17
|
+
/** The scrape's name — every error cause raised inside the session carries it. */
|
|
18
|
+
readonly name: string;
|
|
19
|
+
readonly rules: InterceptRules;
|
|
20
|
+
readonly clock: ScrapeClock;
|
|
21
|
+
/** Per-operation default, in ms. A `waitFor` with its own `timeout` overrides it. */
|
|
22
|
+
readonly timeoutMs: number;
|
|
23
|
+
readonly secrets?: ScrapeSecrets | undefined;
|
|
24
|
+
readonly robots?: RobotsGate | undefined;
|
|
25
|
+
/** The run's cancellation. A driver that ignores it leaves a browser behind on every kill. */
|
|
26
|
+
readonly signal?: AbortSignal | undefined;
|
|
27
|
+
/** A previously captured session, put back before the first navigation. */
|
|
28
|
+
readonly restore?: SessionSnapshot | undefined;
|
|
29
|
+
/** BOTH transports dial through it. A different exit IP mid-session is a different client. */
|
|
30
|
+
readonly proxy?: string | undefined;
|
|
31
|
+
/** Awaited before every navigation AND every HTTP request — one budget across both legs. */
|
|
32
|
+
readonly pace?: ((signal?: AbortSignal) => Promise<void>) | undefined;
|
|
33
|
+
/** Called on every operation on either transport. The wedge watchdog measures the gaps. */
|
|
34
|
+
readonly onActivity?: (() => void) | undefined;
|
|
35
|
+
/**
|
|
36
|
+
* The wedge discipline, as the scrape declared it: kill the browser after `idleMs` of silence,
|
|
37
|
+
* and give a graceful quit `graceMs` before killing it anyway. A driver with no OS process to
|
|
38
|
+
* reach still honours the abort half.
|
|
39
|
+
*/
|
|
40
|
+
readonly watchdog?: { readonly idleMs?: number; readonly graceMs?: number } | undefined;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
export interface ScrapeSession {
|
|
44
|
+
readonly driver: string;
|
|
45
|
+
readonly page: ScrapePage;
|
|
46
|
+
/**
|
|
47
|
+
* The second transport, bound to the SAME session as the page: the browser's cookies, headers
|
|
48
|
+
* and proxy, the same `allowHosts`, the same rate limit, the same cancellation. A driver hands
|
|
49
|
+
* both back together because the two legs of a hybrid scrape are one client, and a session that
|
|
50
|
+
* could hand out only one of them would leave the other to be hand-rolled per app.
|
|
51
|
+
*/
|
|
52
|
+
readonly http: ScrapeHttp;
|
|
53
|
+
/**
|
|
54
|
+
* Ends the session — and for a remote driver that means BOTH halves: the local connection AND
|
|
55
|
+
* the browser somebody else is billing for. A `close()` that only disconnects leaves a paid
|
|
56
|
+
* session running until its provider times it out, which is the bill nobody attributes.
|
|
57
|
+
*
|
|
58
|
+
* Idempotent, and never throws: it runs in a `finally`, and a close that threw would replace
|
|
59
|
+
* the run's real failure with a teardown failure.
|
|
60
|
+
*/
|
|
61
|
+
close(): Promise<void>;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
export interface ScrapeDriver {
|
|
65
|
+
/** `puppeteer` | `fixture` | `fake`. Appears in every error cause raised against it. */
|
|
66
|
+
readonly name: string;
|
|
67
|
+
open(init: SessionInit): Promise<ScrapeSession>;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
let ambient: ScrapeDriver | undefined;
|
|
71
|
+
|
|
72
|
+
/** Set once at boot from `app.config.ts`. A scrape's own `driver:` overrides it per definition. */
|
|
73
|
+
export function setScrapeDriver(driver: ScrapeDriver): void {
|
|
74
|
+
ambient = driver;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
export function scrapeDriver(): ScrapeDriver | undefined {
|
|
78
|
+
return ambient;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Test/CLI seam: forget the ambient driver. The counterpart to `resetJobDriver()` — a test that
|
|
83
|
+
* installs a browser has to be able to put the process back, or every later file in the same bun
|
|
84
|
+
* process runs against a driver it never asked for.
|
|
85
|
+
*/
|
|
86
|
+
export function resetScrapeDriver(): void {
|
|
87
|
+
ambient = undefined;
|
|
88
|
+
}
|