@ultimat3/scraping 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,47 @@
1
+ // Reading a thrown value safely. Every classification decision this package makes — is this
2
+ // terminal, does this burn the session, what code does the event line carry — asks these two
3
+ // questions, and both have to be answerable about a value nobody here constructed.
4
+
5
+ import { isUltimateError } from '@ultimat3/core';
6
+
7
+ /** The stable code, or `undefined`. Never the message: a message is whatever a site put in it. */
8
+ export function errorCode(thrown: unknown): string | undefined {
9
+ if (isUltimateError(thrown)) return thrown.code;
10
+ // A structural read, not `instanceof`: an error crossing a worker or a subprocess arrives as a
11
+ // plain object, and refusing to recognise it there is how a terminal failure gets retried.
12
+ if (typeof thrown === 'object' && thrown !== null) {
13
+ const code: unknown = (thrown as { code?: unknown }).code;
14
+ if (typeof code === 'string' && code.startsWith('X_')) return code;
15
+ }
16
+ return undefined;
17
+ }
18
+
19
+ /**
20
+ * Codes after which the persisted identity must not be reused. `X_SCRAPE_BLOCKED` is the whole
21
+ * reason this list exists: a flagged profile stays flagged, so the retry has to arrive as
22
+ * somebody else or it fails identically, forever.
23
+ */
24
+ export const BURNS_SESSION: ReadonlySet<string> = new Set([
25
+ 'X_SCRAPE_BLOCKED',
26
+ 'X_SCRAPE_SESSION_EXPIRED',
27
+ ]);
28
+
29
+ /**
30
+ * Codes no retry may ever follow, whatever the retry policy says. `X_SCRAPE_AUTH_FAILED` is the
31
+ * one that matters: a site that locks an account after three wrong attempts turns a retrying
32
+ * framework into the thing that destroys the user's account.
33
+ */
34
+ export const NEVER_RETRIED: ReadonlySet<string> = new Set([
35
+ 'X_SCRAPE_AUTH_FAILED',
36
+ 'X_SCRAPE_PROMPT_UNANSWERED',
37
+ ]);
38
+
39
+ export const burnsSession = (thrown: unknown): boolean => {
40
+ const code = errorCode(thrown);
41
+ return code !== undefined && BURNS_SESSION.has(code);
42
+ };
43
+
44
+ export const neverRetried = (thrown: unknown): boolean => {
45
+ const code = errorCode(thrown);
46
+ return code !== undefined && NEVER_RETRIED.has(code);
47
+ };
package/src/hosts.ts ADDED
@@ -0,0 +1,56 @@
1
+ // `allowHosts`, as a decision every driver asks before a request leaves — never a note in a
2
+ // README. A headless browser inside your network is the widest SSRF surface an app can own: one
3
+ // injected `<img src="http://169.254.169.254/…">` on a page you do not control is a credential
4
+ // read, and no amount of "we only visit example.com" in prose intercepts it.
5
+
6
+ export type HostRule = string;
7
+
8
+ /** The one spelling that means "every host", written out so it is visible in review. */
9
+ export const ANY_HOST: HostRule = '*';
10
+
11
+ /**
12
+ * Schemes with no host to match. `about:blank` is where every browser starts, `data:` and `blob:`
13
+ * never leave the process — refusing them would refuse the first page load of every run.
14
+ */
15
+ const HOSTLESS_SCHEMES = new Set(['about:', 'data:', 'blob:', 'javascript:']);
16
+
17
+ export interface HostDecision {
18
+ readonly allowed: boolean;
19
+ /** The host the URL resolved to, `''` for a hostless scheme. */
20
+ readonly host: string;
21
+ }
22
+
23
+ /**
24
+ * `example.com` matches that host EXACTLY. `*.example.com` matches any subdomain and NOT the
25
+ * apex — the two are written separately on purpose: an allow list that silently included every
26
+ * subdomain would let a `cdn-user-content.example.com` (whose contents somebody else controls)
27
+ * through a rule an author wrote for the apex.
28
+ */
29
+ export function hostMatches(host: string, rule: HostRule): boolean {
30
+ if (rule === ANY_HOST) return true;
31
+ const normalised = host.toLowerCase();
32
+ const cleaned = rule.trim().toLowerCase();
33
+ if (cleaned.startsWith('*.')) {
34
+ const suffix = cleaned.slice(1);
35
+ return normalised.endsWith(suffix) && normalised.length > suffix.length;
36
+ }
37
+ return normalised === cleaned;
38
+ }
39
+
40
+ /**
41
+ * Fails CLOSED: a URL that cannot be parsed is refused. A driver handed a malformed request has
42
+ * no way to know where it would have gone, and "we could not tell, so we let it through" is the
43
+ * decision that makes the whole list advisory.
44
+ */
45
+ export function hostDecision(url: string, allowHosts: readonly HostRule[]): HostDecision {
46
+ const scheme = url.slice(0, Math.max(0, url.indexOf(':') + 1)).toLowerCase();
47
+ if (HOSTLESS_SCHEMES.has(scheme)) return { allowed: true, host: '' };
48
+ let host: string;
49
+ try {
50
+ host = new URL(url).hostname;
51
+ } catch {
52
+ return { allowed: false, host: '' };
53
+ }
54
+ if (host === '') return { allowed: false, host };
55
+ return { allowed: allowHosts.some((rule) => hostMatches(host, rule)), host };
56
+ }
@@ -0,0 +1,119 @@
1
+ // CSS selectors over an HTML string, on Bun's own `HTMLRewriter` — no jsdom, no cheerio, no
2
+ // dependency at all. This is what makes the fake and the fixture drivers real enough to be worth
3
+ // testing against: they answer the same `query()` the browser driver does, from markup, offline.
4
+ //
5
+ // The selector subset is `HTMLRewriter`'s (lol-html): type, `#id`, `.class`, `[attr]`,
6
+ // `[attr=value]`, descendant and child combinators. A selector it refuses is refused loudly.
7
+
8
+ import type { ElementSnapshot } from './target';
9
+ import { ROOT_SELECTOR } from './target';
10
+
11
+ /** Elements with no end tag — `onEndTag` is not available for them, so they close on open. */
12
+ const VOID_TAGS = new Set([
13
+ 'area',
14
+ 'base',
15
+ 'br',
16
+ 'col',
17
+ 'embed',
18
+ 'hr',
19
+ 'img',
20
+ 'input',
21
+ 'link',
22
+ 'meta',
23
+ 'source',
24
+ 'track',
25
+ 'wbr',
26
+ ]);
27
+
28
+ const HIDDEN_STYLE = /(?:^|;)\s*(?:display\s*:\s*none|visibility\s*:\s*hidden)\s*(?:;|$)/i;
29
+
30
+ /** What "rendered" can mean without a layout engine: the markup says so, or it does not. */
31
+ export const markupVisible = (tag: string, attrs: Readonly<Record<string, string>>): boolean => {
32
+ if ('hidden' in attrs) return false;
33
+ if (tag === 'input' && attrs['type']?.toLowerCase() === 'hidden') return false;
34
+ const style = attrs['style'];
35
+ if (style !== undefined && HIDDEN_STYLE.test(style)) return false;
36
+ return attrs['aria-hidden'] !== 'true';
37
+ };
38
+
39
+ export const markupEnabled = (attrs: Readonly<Record<string, string>>): boolean =>
40
+ !('disabled' in attrs) && attrs['aria-disabled'] !== 'true';
41
+
42
+ interface Open {
43
+ readonly tag: string;
44
+ readonly attrs: Record<string, string>;
45
+ readonly parts: string[];
46
+ }
47
+
48
+ const finish = (open: Open): ElementSnapshot => ({
49
+ tag: open.tag,
50
+ attrs: open.attrs,
51
+ text: open.parts.join('').replaceAll(/\s+/g, ' ').trim(),
52
+ value: open.attrs['value'] ?? '',
53
+ visible: markupVisible(open.tag, open.attrs),
54
+ enabled: markupEnabled(open.attrs),
55
+ // No box and no hit-target: THIS TARGET HAS NO LAYOUT ENGINE. Fabricating either is how an
56
+ // offline suite reports a covered button as clickable. See `actionability.ts`.
57
+ });
58
+
59
+ /**
60
+ * Every element matching `selector`, in document order, with its text content flattened.
61
+ *
62
+ * `ROOT_SELECTOR` answers even for a fragment with no `<html>` element — a fixture is usually a
63
+ * body snippet, and `page.text()` returning `''` for one would be a silent wrong answer rather
64
+ * than a missing element.
65
+ */
66
+ export async function queryHtml(
67
+ html: string,
68
+ selector: string,
69
+ ): Promise<readonly ElementSnapshot[]> {
70
+ const done: ElementSnapshot[] = [];
71
+ const open: Open[] = [];
72
+ const allText: string[] = [];
73
+ const rewriter = new HTMLRewriter()
74
+ .on('*', {
75
+ text(chunk): void {
76
+ allText.push(chunk.text);
77
+ },
78
+ })
79
+ .on(selector, {
80
+ element(element): void {
81
+ const tag = element.tagName.toLowerCase();
82
+ const attrs: Record<string, string> = {};
83
+ for (const [name, value] of element.attributes) attrs[name.toLowerCase()] = value;
84
+ const record: Open = { tag, attrs, parts: [] };
85
+ if (VOID_TAGS.has(tag)) {
86
+ done.push(finish(record));
87
+ return;
88
+ }
89
+ open.push(record);
90
+ element.onEndTag(() => {
91
+ const index = open.lastIndexOf(record);
92
+ if (index !== -1) open.splice(index, 1);
93
+ done.push(finish(record));
94
+ });
95
+ },
96
+ text(chunk): void {
97
+ // Every OPEN match, not just the innermost: a `<div>` and the `<span>` inside it both
98
+ // legitimately match `div, span`, and both contain the text.
99
+ for (const record of open) record.parts.push(chunk.text);
100
+ },
101
+ });
102
+ await rewriter.transform(new Response(html)).text();
103
+ // Anything still open never saw its end tag (malformed markup). It is still an element that
104
+ // matched, so it is reported rather than dropped — a scraper's input is somebody else's HTML.
105
+ for (const record of open) done.push(finish(record));
106
+ if (done.length === 0 && selector === ROOT_SELECTOR) {
107
+ return [
108
+ {
109
+ tag: 'html',
110
+ attrs: {},
111
+ text: allText.join('').replaceAll(/\s+/g, ' ').trim(),
112
+ value: '',
113
+ visible: true,
114
+ enabled: true,
115
+ },
116
+ ];
117
+ }
118
+ return done;
119
+ }
@@ -0,0 +1,45 @@
1
+ // The subresource requests a document would make, read out of its markup. The offline drivers
2
+ // have no network stack, so this is how they still exercise interception: an `<img>` pointing at
3
+ // a host `allowHosts` does not list produces the same refused network entry offline that a real
4
+ // browser produces on the wire.
5
+
6
+ import { queryHtml } from './html-query';
7
+ import type { ResourceType } from './rings';
8
+
9
+ export interface MarkupRequest {
10
+ readonly url: string;
11
+ readonly resourceType: ResourceType;
12
+ }
13
+
14
+ /** Selector, attribute and the type a browser would classify it as. `<a href>` is not a request. */
15
+ const SOURCES: readonly (readonly [string, string, ResourceType])[] = [
16
+ ['img', 'src', 'image'],
17
+ ['script', 'src', 'script'],
18
+ ['link', 'href', 'stylesheet'],
19
+ ['iframe', 'src', 'document'],
20
+ ['source', 'src', 'media'],
21
+ ['video', 'src', 'media'],
22
+ ['audio', 'src', 'media'],
23
+ ];
24
+
25
+ /** Absolute URLs only: a relative one resolves against `base`, which is the page's own URL. */
26
+ export async function markupRequests(
27
+ html: string,
28
+ base: string,
29
+ ): Promise<readonly MarkupRequest[]> {
30
+ const requests: MarkupRequest[] = [];
31
+ for (const [selector, attribute, resourceType] of SOURCES) {
32
+ for (const element of await queryHtml(html, selector)) {
33
+ const raw = element.attrs[attribute];
34
+ if (raw === undefined || raw === '') continue;
35
+ // `<link>` covers icons, preloads and manifests too; only a stylesheet is a stylesheet.
36
+ if (selector === 'link' && (element.attrs['rel'] ?? '') !== 'stylesheet') continue;
37
+ try {
38
+ requests.push({ url: new URL(raw, base).toString(), resourceType });
39
+ } catch {
40
+ // A specifier no URL parser accepts is not a request any browser would make either.
41
+ }
42
+ }
43
+ }
44
+ return requests;
45
+ }
@@ -0,0 +1,229 @@
1
+ // A `ScrapeTarget` over recorded HTML: no process, no port, no CDP. This is what `bun test` runs
2
+ // against, and it is the reason a scraper's tests need no Chrome.
3
+ //
4
+ // The rule that makes it worth having: an UNRECORDED request THROWS. A driver that quietly fell
5
+ // through to the network would make a green offline suite that is secretly hitting production —
6
+ // the exact failure an offline driver exists to prevent.
7
+
8
+ import type { ScrapeClock } from './clock';
9
+ import { browserUnreachable, downloadTimeout, fixtureMissing, fixtureStale } from './error-throws';
10
+ import { queryHtml } from './html-query';
11
+ import { markupRequests } from './html-requests';
12
+ import type { InterceptRules } from './intercept';
13
+ import { interceptVerdict, refusalEntry } from './intercept';
14
+ import type { PageRecording } from './recording';
15
+ import { splitDownload } from './recording';
16
+ import type { ConsoleLine, NetworkEntry } from './rings';
17
+ import { createRing } from './rings';
18
+ import type { SessionSnapshot } from './session-state';
19
+ import { EMPTY_SESSION } from './session-state';
20
+ import type {
21
+ CaptureOptions,
22
+ ElementSnapshot,
23
+ FrameRef,
24
+ GotoOptions,
25
+ ScrapeCookie,
26
+ ScrapeDownloadFile,
27
+ ScrapeTarget,
28
+ } from './target';
29
+
30
+ /** Deterministic bytes, so an artifact test asserts on a stable digest. Not a real PNG render. */
31
+ const FAKE_PNG = new Uint8Array([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a]);
32
+ const FAKE_PDF = new TextEncoder().encode('%PDF-1.4 offline');
33
+
34
+ export type RecordingLookup = (url: string) => Promise<PageRecording | undefined>;
35
+
36
+ export interface HtmlTargetInit {
37
+ readonly driver: string;
38
+ readonly lookup: RecordingLookup;
39
+ readonly rules: InterceptRules;
40
+ readonly clock: ScrapeClock;
41
+ /** Where recordings come from, for the cause line: a directory, or `fakeBrowser()`. */
42
+ readonly source: string;
43
+ readonly start?: PageRecording | undefined;
44
+ readonly maxAgeMs?: number | undefined;
45
+ readonly cookies?: readonly ScrapeCookie[] | undefined;
46
+ /** What `page.session()` answers, so a fixture can assert on the browser-to-HTTP handoff. */
47
+ readonly session?: SessionSnapshot | undefined;
48
+ }
49
+
50
+ const EMPTY: PageRecording = { url: 'about:blank', html: '' };
51
+
52
+ /** Typed text is an overlay keyed by `id`, then `name`, then the selector used to type it. */
53
+ const keyOf = (selector: string, element: ElementSnapshot | undefined): string => {
54
+ if (element === undefined) return selector;
55
+ const id = element.attrs['id'];
56
+ if (id !== undefined) return `#${id}`;
57
+ const name = element.attrs['name'];
58
+ if (name !== undefined) return `[name=${name}]`;
59
+ return selector;
60
+ };
61
+
62
+ export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
63
+ const consoleRing = createRing<ConsoleLine>();
64
+ const networkRing = createRing<NetworkEntry>();
65
+ const overlay = new Map<string, string>();
66
+ let page: PageRecording = init.start ?? EMPTY;
67
+ let armed: string | undefined;
68
+ let closed = false;
69
+ let session: SessionSnapshot = init.session ?? {
70
+ ...EMPTY_SESSION,
71
+ cookies: init.cookies ?? [],
72
+ };
73
+
74
+ const live = (): void => {
75
+ if (closed) throw browserUnreachable(init.driver, 'the offline target is already closed');
76
+ };
77
+
78
+ const withOverlay = (selector: string, elements: readonly ElementSnapshot[]): ElementSnapshot[] =>
79
+ elements.map((element) => {
80
+ const typed = overlay.get(keyOf(selector, element));
81
+ return typed === undefined ? element : { ...element, value: typed };
82
+ });
83
+
84
+ const query = async (selector: string): Promise<readonly ElementSnapshot[]> => {
85
+ live();
86
+ return withOverlay(selector, await queryHtml(page.html, selector));
87
+ };
88
+
89
+ const at = (selector: string, index: number): Promise<ElementSnapshot | undefined> =>
90
+ query(selector).then((elements) => elements[index]);
91
+
92
+ /** Interception, offline: every request the markup would make, judged by the same rule. */
93
+ const intercept = async (recording: PageRecording): Promise<void> => {
94
+ for (const request of await markupRequests(recording.html, recording.url)) {
95
+ const verdict = interceptVerdict(request.url, request.resourceType, init.rules);
96
+ const now = init.clock.now().getTime();
97
+ networkRing.push(
98
+ verdict === 'allow'
99
+ ? {
100
+ method: 'GET',
101
+ url: request.url,
102
+ resourceType: request.resourceType,
103
+ at: now,
104
+ status: 200,
105
+ }
106
+ : refusalEntry(request.url, request.resourceType, verdict, now),
107
+ );
108
+ }
109
+ };
110
+
111
+ const load = async (url: string): Promise<void> => {
112
+ const found = await init.lookup(url);
113
+ if (found === undefined) throw fixtureMissing(url, init.source);
114
+ if (init.maxAgeMs !== undefined && found.recordedAt !== undefined) {
115
+ const age = init.clock.now().getTime() - new Date(found.recordedAt).getTime();
116
+ if (age > init.maxAgeMs) throw fixtureStale(url, age, init.maxAgeMs);
117
+ }
118
+ page = found;
119
+ overlay.clear();
120
+ armed = undefined;
121
+ networkRing.push({
122
+ method: 'GET',
123
+ url,
124
+ resourceType: 'document',
125
+ status: 200,
126
+ at: init.clock.now().getTime(),
127
+ });
128
+ await intercept(found);
129
+ };
130
+
131
+ /** Relative specifiers resolve against the current page, exactly as a browser resolves them. */
132
+ const navigate = async (url: string): Promise<void> => {
133
+ live();
134
+ let absolute: string;
135
+ try {
136
+ absolute = new URL(url, page.url === 'about:blank' ? undefined : page.url).toString();
137
+ } catch {
138
+ throw fixtureMissing(url, init.source);
139
+ }
140
+ await load(absolute);
141
+ };
142
+
143
+ const frameTarget = (html: string, url: string): ScrapeTarget => ({
144
+ ...base,
145
+ url: () => url,
146
+ content: () => Promise.resolve(html),
147
+ query: async (selector) => withOverlay(selector, await queryHtml(html, selector)),
148
+ frames: () => Promise.resolve([]),
149
+ });
150
+
151
+ const base: ScrapeTarget = {
152
+ driver: init.driver,
153
+ console: consoleRing,
154
+ network: networkRing,
155
+ url: () => page.url,
156
+ goto: (url: string, _options: GotoOptions): Promise<void> => navigate(url),
157
+ content: (): Promise<string> => {
158
+ live();
159
+ return Promise.resolve(page.html);
160
+ },
161
+ query,
162
+ async click(selector: string, index: number): Promise<void> {
163
+ const element = await at(selector, index);
164
+ if (element === undefined) throw fixtureMissing(`${page.url} ${selector}`, init.source);
165
+ const download = page.downloads?.[selector] ?? page.downloads?.[element.attrs['id'] ?? ''];
166
+ if (download !== undefined) armed = download;
167
+ const href =
168
+ element.attrs['data-goto'] ?? (element.tag === 'a' ? element.attrs['href'] : undefined);
169
+ if (href !== undefined && href !== '') await navigate(href);
170
+ },
171
+ async type(selector: string, text: string): Promise<void> {
172
+ const element = await at(selector, 0);
173
+ const key = keyOf(selector, element);
174
+ overlay.set(key, `${overlay.get(key) ?? element?.value ?? ''}${text}`);
175
+ },
176
+ async clear(selector: string): Promise<void> {
177
+ overlay.set(keyOf(selector, await at(selector, 0)), '');
178
+ },
179
+ async select(selector: string, values: readonly string[]): Promise<void> {
180
+ overlay.set(keyOf(selector, await at(selector, 0)), values[0] ?? '');
181
+ },
182
+ evaluate(expression: string): Promise<unknown> {
183
+ live();
184
+ const recorded = page.evaluate?.[expression];
185
+ // Unrecorded and therefore refused, for the same reason an unrecorded page is: an offline
186
+ // driver that invented an answer here would make the assertion above it meaningless.
187
+ if (recorded === undefined)
188
+ throw fixtureMissing(`${page.url} evaluate(${expression})`, init.source);
189
+ return Promise.resolve(JSON.parse(recorded) as unknown);
190
+ },
191
+ screenshot: (_options: CaptureOptions): Promise<Uint8Array> => Promise.resolve(FAKE_PNG),
192
+ pdf: (_options: CaptureOptions): Promise<Uint8Array> => Promise.resolve(FAKE_PDF),
193
+ cookies: (): Promise<readonly ScrapeCookie[]> => Promise.resolve(session.cookies),
194
+ session: (): Promise<SessionSnapshot> => Promise.resolve(session),
195
+ restore: (next: SessionSnapshot): Promise<void> => {
196
+ session = next;
197
+ return Promise.resolve();
198
+ },
199
+ download(options: { readonly timeoutMs: number }): Promise<ScrapeDownloadFile> {
200
+ if (armed === undefined) throw downloadTimeout(options.timeoutMs, page.url);
201
+ const { filename, contents } = splitDownload(armed);
202
+ armed = undefined;
203
+ return Promise.resolve({ filename, bytes: new TextEncoder().encode(contents) });
204
+ },
205
+ async frames(): Promise<readonly FrameRef[]> {
206
+ const refs: FrameRef[] = [];
207
+ for (const element of await queryHtml(page.html, 'iframe')) {
208
+ const name = element.attrs['name'] ?? element.attrs['id'] ?? '';
209
+ const src = element.attrs['src'] ?? '';
210
+ const html = page.frames?.[name] ?? page.frames?.[src];
211
+ if (html === undefined)
212
+ throw fixtureMissing(`${page.url} iframe ${name || src}`, init.source);
213
+ const url = src === '' ? page.url : new URL(src, page.url).toString();
214
+ refs.push({
215
+ name,
216
+ url,
217
+ selector: element.attrs['id'] === undefined ? undefined : `#${element.attrs['id']}`,
218
+ target: frameTarget(html, url),
219
+ });
220
+ }
221
+ return refs;
222
+ },
223
+ close(): Promise<void> {
224
+ closed = true;
225
+ return Promise.resolve();
226
+ },
227
+ };
228
+ return base;
229
+ }
@@ -0,0 +1,85 @@
1
+ // The HTTP leg, offline — the second half of the ONE fixture format. A hybrid scrape replays end
2
+ // to end from a single directory: browser login, session handoff, HTTP bulk fetch.
3
+ //
4
+ // Same rule as the page half: an unrecorded request THROWS. A transport that fell through to the
5
+ // real network would make the hybrid path — which is where the interesting code lives — the one
6
+ // path a test never really covers.
7
+
8
+ import type { ScrapeClock } from './clock';
9
+ import { fixtureMissing, fixtureStale, hostBlocked } from './error-throws';
10
+ import type { HttpRequestInit, ScrapeHttp, ScrapeResponse } from './http';
11
+ import { responseOver } from './http';
12
+ import type { InterceptRules } from './intercept';
13
+ import { interceptVerdict } from './intercept';
14
+ import type { HttpRecording } from './recording';
15
+ import type { NetworkRing } from './rings';
16
+ import type { RobotsGate } from './robots';
17
+
18
+ export type HttpRecordingLookup = (
19
+ method: string,
20
+ url: string,
21
+ ) => Promise<HttpRecording | undefined>;
22
+
23
+ export interface RecordedHttpInit {
24
+ readonly lookup: HttpRecordingLookup;
25
+ readonly rules: InterceptRules;
26
+ readonly network: NetworkRing;
27
+ readonly clock: ScrapeClock;
28
+ readonly source: string;
29
+ /**
30
+ * The SAME gate the page leg holds. Without it a hybrid scrape whose JSON endpoint is
31
+ * `Disallow:`ed replayed green offline and threw terminal `X_SCRAPE_ROBOTS_DISALLOWED` on its
32
+ * first real attempt — the identical argument the host rule below already makes for itself.
33
+ */
34
+ readonly robots?: RobotsGate | undefined;
35
+ readonly maxAgeMs?: number | undefined;
36
+ }
37
+
38
+ /** `GET https://api.example.com/v1/orders?page=2` -> `http-get-api-example-com-v1-orders-page-2`. */
39
+ export function httpRecordingFilename(method: string, url: string): string {
40
+ const slug = `${method}-${url.replace(/^[a-z]+:\/\//i, '')}`
41
+ .toLowerCase()
42
+ .replaceAll(/[^a-z0-9]+/g, '-')
43
+ .replace(/^-+|-+$/g, '');
44
+ return `http-${slug}.json`;
45
+ }
46
+
47
+ export function recordedHttp(init: RecordedHttpInit): ScrapeHttp {
48
+ return {
49
+ async request(url: string, request: HttpRequestInit = {}): Promise<ScrapeResponse> {
50
+ const method = (request.method ?? 'GET').toUpperCase();
51
+ // The host rule applies to the offline leg too. Otherwise a scrape that a fixture proves
52
+ // correct would be the first thing to hit a host `allowHosts` forbids, in production.
53
+ if (interceptVerdict(url, 'fetch', init.rules) !== 'allow') {
54
+ throw hostBlocked(url, init.rules.allowHosts);
55
+ }
56
+ await init.robots?.assertAllowed(url);
57
+ const found = await init.lookup(method, url);
58
+ if (found === undefined) throw fixtureMissing(`${method} ${url}`, init.source);
59
+ if (init.maxAgeMs !== undefined && found.recordedAt !== undefined) {
60
+ const age = init.clock.now().getTime() - new Date(found.recordedAt).getTime();
61
+ if (age > init.maxAgeMs) throw fixtureStale(`${method} ${url}`, age, init.maxAgeMs);
62
+ }
63
+ init.network.push({
64
+ method,
65
+ url,
66
+ status: found.status,
67
+ resourceType: 'fetch',
68
+ at: init.clock.now().getTime(),
69
+ });
70
+ return responseOver(url, found.status, found.headers ?? {}, () =>
71
+ Promise.resolve(found.body),
72
+ );
73
+ },
74
+ };
75
+ }
76
+
77
+ /** In-memory recordings, keyed `METHOD url`. What `fakeBrowser({ http })` is built from. */
78
+ export function httpRecordingsOf(recordings: readonly HttpRecording[]): Map<string, HttpRecording> {
79
+ return new Map(
80
+ recordings.map((recording) => [
81
+ `${recording.method.toUpperCase()} ${recording.url}`,
82
+ recording,
83
+ ]),
84
+ );
85
+ }