@ultimat3/scraping 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +194 -0
- package/package.json +38 -0
- package/src/actionability.ts +106 -0
- package/src/artifacts.ts +69 -0
- package/src/auth.ts +200 -0
- package/src/cdp-fake.ts +150 -0
- package/src/cdp-port.ts +75 -0
- package/src/cdp-snapshot.ts +63 -0
- package/src/cdp-target.ts +320 -0
- package/src/clock.ts +97 -0
- package/src/cookie-scope.ts +97 -0
- package/src/driver-cdp.ts +184 -0
- package/src/driver-fake.ts +119 -0
- package/src/driver-fixture.ts +65 -0
- package/src/driver.ts +88 -0
- package/src/error-throws.ts +258 -0
- package/src/errors.ts +180 -0
- package/src/events.ts +74 -0
- package/src/expect.ts +133 -0
- package/src/failures.ts +47 -0
- package/src/hosts.ts +56 -0
- package/src/html-query.ts +119 -0
- package/src/html-requests.ts +45 -0
- package/src/html-target.ts +229 -0
- package/src/http-recorded.ts +85 -0
- package/src/http.ts +158 -0
- package/src/index.ts +178 -0
- package/src/intercept.ts +41 -0
- package/src/offline-session.ts +67 -0
- package/src/page-over-target.ts +215 -0
- package/src/page.ts +103 -0
- package/src/rate.ts +23 -0
- package/src/recording.ts +68 -0
- package/src/recover.ts +51 -0
- package/src/rings.ts +73 -0
- package/src/robots.ts +144 -0
- package/src/scrape-run.ts +226 -0
- package/src/scrape.ts +151 -0
- package/src/secrets.ts +91 -0
- package/src/session-state.ts +181 -0
- package/src/target.ts +118 -0
- package/src/watchdog.ts +100 -0
package/src/page.ts
ADDED
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
// The vocabulary a `run()` body writes against — small, declarative, and driver-blind. Fourteen
|
|
2
|
+
// verbs, chosen because every scraper in the audit re-implemented these and nothing else; a
|
|
3
|
+
// fifteenth would be a second way to do something on this list.
|
|
4
|
+
//
|
|
5
|
+
// Nothing here mentions puppeteer, CDP, a frame handle or a locator. That is the seam: a run body
|
|
6
|
+
// written against this file runs unchanged on a real browser, on a recorded fixture and on a
|
|
7
|
+
// parsed HTML string, which is what makes `bun test` need no Chrome.
|
|
8
|
+
|
|
9
|
+
import type { Secret } from '@ultimat3/core';
|
|
10
|
+
import type { ActionabilityState } from './actionability';
|
|
11
|
+
import type { ConsoleLine, NetworkEntry } from './rings';
|
|
12
|
+
import type { SessionSnapshot } from './session-state';
|
|
13
|
+
import type { ElementSnapshot, ScrapeCookie, ScrapeDownloadFile } from './target';
|
|
14
|
+
|
|
15
|
+
export interface WaitOptions {
|
|
16
|
+
readonly state?: ActionabilityState | undefined;
|
|
17
|
+
/** Milliseconds. Falls back to the scrape's own `timeout`. */
|
|
18
|
+
readonly timeout?: number | undefined;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
export interface ElementValue {
|
|
22
|
+
readonly tag: string;
|
|
23
|
+
readonly text: string;
|
|
24
|
+
/** The control's value, `''` for anything that is not one. */
|
|
25
|
+
readonly value: string;
|
|
26
|
+
readonly attrs: Readonly<Record<string, string>>;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* A frame, or the document — everything both can do. Held as a LAZY handle: every method
|
|
31
|
+
* re-resolves its underlying frame at call time, so a handle taken before a navigation still
|
|
32
|
+
* addresses the right frame afterwards rather than a detached one.
|
|
33
|
+
*/
|
|
34
|
+
export interface ScrapeFrame {
|
|
35
|
+
url(): string;
|
|
36
|
+
/** Blocks until the element reaches `state` (default `actionable`), then answers its snapshot. */
|
|
37
|
+
waitFor(selector: string, options?: WaitOptions): Promise<ElementSnapshot>;
|
|
38
|
+
/**
|
|
39
|
+
* Waits for the element to be visible, enabled, unobstructed and STILL before clicking. A raw
|
|
40
|
+
* driver's click waits for the selector alone, which is why every app that uses one grows a
|
|
41
|
+
* `waitForTimeout(800)` somewhere above it.
|
|
42
|
+
*/
|
|
43
|
+
click(selector: string, options?: WaitOptions): Promise<void>;
|
|
44
|
+
/** Appends. A `Secret` marks this page as carrying one, which refuses later pixel captures. */
|
|
45
|
+
type(selector: string, text: string | Secret, options?: WaitOptions): Promise<void>;
|
|
46
|
+
/** Clears first, then types — the spelling a login form wants. */
|
|
47
|
+
fill(selector: string, text: string | Secret, options?: WaitOptions): Promise<void>;
|
|
48
|
+
select(selector: string, values: readonly string[], options?: WaitOptions): Promise<void>;
|
|
49
|
+
/** Every match, as values. Row assembly is the app's business, never the framework's. */
|
|
50
|
+
values(selector: string): Promise<readonly ElementValue[]>;
|
|
51
|
+
/** The first match's text, or `''`. */
|
|
52
|
+
text(selector?: string): Promise<string>;
|
|
53
|
+
/** Serialised HTML, redacted by value and with password fields blanked. */
|
|
54
|
+
html(): Promise<string>;
|
|
55
|
+
count(selector: string): Promise<number>;
|
|
56
|
+
/** The expression's result, `unknown` — parse it with a schema, never cast it. */
|
|
57
|
+
evaluate(expression: string): Promise<unknown>;
|
|
58
|
+
/**
|
|
59
|
+
* A child frame by name, `id`, or `<iframe>` selector. RE-RESOLVED on every call made through
|
|
60
|
+
* the returned handle: a frame handle captured before a re-navigation is the single biggest
|
|
61
|
+
* correctness trap in this whole vocabulary, and the type cannot express "stale" — so nothing
|
|
62
|
+
* here holds one.
|
|
63
|
+
*/
|
|
64
|
+
frame(nameOrSelector: string): ScrapeFrame;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
export interface CaptureRequest {
|
|
68
|
+
readonly fullPage?: boolean | undefined;
|
|
69
|
+
readonly timeout?: number | undefined;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
export interface DownloadRequest {
|
|
73
|
+
readonly timeout?: number | undefined;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
export interface ScrapePage extends ScrapeFrame {
|
|
77
|
+
/**
|
|
78
|
+
* Navigate. Refused before a byte leaves when the host is not in `allowHosts`
|
|
79
|
+
* (`X_SCRAPE_HOST_BLOCKED`) or robots.txt disallows the path (`X_SCRAPE_ROBOTS_DISALLOWED`).
|
|
80
|
+
*/
|
|
81
|
+
goto(url: string, options?: { readonly timeout?: number | undefined }): Promise<void>;
|
|
82
|
+
/** PNG bytes. REFUSED once a secret has been typed into this page — see `secrets.ts`. */
|
|
83
|
+
screenshot(options?: CaptureRequest): Promise<Uint8Array>;
|
|
84
|
+
/** PDF bytes. Refused on the same condition, for the same reason. */
|
|
85
|
+
pdf(options?: CaptureRequest): Promise<Uint8Array>;
|
|
86
|
+
/** The file the last click produced, or `X_SCRAPE_DOWNLOAD_TIMEOUT`. */
|
|
87
|
+
download(options?: DownloadRequest): Promise<ScrapeDownloadFile>;
|
|
88
|
+
cookies(): Promise<readonly ScrapeCookie[]>;
|
|
89
|
+
/**
|
|
90
|
+
* The handoff, made explicit: what the HTTP leg will send, as a value an author can inspect and
|
|
91
|
+
* a fixture can assert on. `http` uses it automatically — this is for seeing what carried over.
|
|
92
|
+
*/
|
|
93
|
+
session(): Promise<SessionSnapshot>;
|
|
94
|
+
/** The bounded tail. Bounded because a long run's full history is an OOM, not a log. */
|
|
95
|
+
console(): readonly ConsoleLine[];
|
|
96
|
+
network(): readonly NetworkEntry[];
|
|
97
|
+
/**
|
|
98
|
+
* How many entries the bound above threw away. It is the same honesty `Ring.dropped` carries:
|
|
99
|
+
* any count taken from `network()` — the run's `refused` total included — is a floor once this
|
|
100
|
+
* is non-zero, and a scrape that blocked 5,000 images otherwise reports 200 with no hint.
|
|
101
|
+
*/
|
|
102
|
+
networkDropped(): number;
|
|
103
|
+
}
|
package/src/rate.ts
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
// Navigations per second, paced against the injected clock. A scraper with no pacing is a
|
|
2
|
+
// scraper that looks like a denial of service to the site it depends on, and the ban is charged
|
|
3
|
+
// to whoever owns the IP — usually the whole fleet, not the one run.
|
|
4
|
+
|
|
5
|
+
import type { ScrapeClock } from './clock';
|
|
6
|
+
|
|
7
|
+
/** Deliberately slow. Raising it is a decision; there is no way to turn pacing off. */
|
|
8
|
+
export const DEFAULT_NAVIGATION_RATE = 1;
|
|
9
|
+
|
|
10
|
+
export type Pacer = (signal?: AbortSignal) => Promise<void>;
|
|
11
|
+
|
|
12
|
+
export function createPacer(rate: number, clock: ScrapeClock): Pacer {
|
|
13
|
+
const intervalMs = 1_000 / rate;
|
|
14
|
+
let nextAt = 0;
|
|
15
|
+
return async (signal?: AbortSignal): Promise<void> => {
|
|
16
|
+
const now = clock.monotonic();
|
|
17
|
+
const waitMs = Math.max(0, nextAt - now);
|
|
18
|
+
// Booked BEFORE the wait, so two concurrent navigations queue behind each other instead of
|
|
19
|
+
// both reading the same free slot and leaving together.
|
|
20
|
+
nextAt = Math.max(now, nextAt) + intervalMs;
|
|
21
|
+
if (waitMs > 0) await clock.sleep(waitMs, signal);
|
|
22
|
+
};
|
|
23
|
+
}
|
package/src/recording.ts
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
// What an offline driver replays: one page, as data. The same shape whether it was written by
|
|
2
|
+
// hand for `fakeBrowser()` or read off disk by `fixtureBrowser()`, so a test that outgrows an
|
|
3
|
+
// inline fixture moves to a directory without a single edit to the run body.
|
|
4
|
+
|
|
5
|
+
import type { StandardSchemaV1 } from '@ultimat3/schema';
|
|
6
|
+
import { parse, t } from '@ultimat3/schema';
|
|
7
|
+
|
|
8
|
+
export interface PageRecording {
|
|
9
|
+
readonly url: string;
|
|
10
|
+
readonly html: string;
|
|
11
|
+
/**
|
|
12
|
+
* `expression -> JSON`. `evaluate()` answers `unknown` on every driver, so a recording holds
|
|
13
|
+
* the JSON text and the target parses it — an expression with no entry is an UNRECORDED
|
|
14
|
+
* request and throws, exactly like an unrecorded navigation.
|
|
15
|
+
*/
|
|
16
|
+
readonly evaluate?: Readonly<Record<string, string>> | undefined;
|
|
17
|
+
/** `<iframe>` name or `src` -> that frame's HTML. */
|
|
18
|
+
readonly frames?: Readonly<Record<string, string>> | undefined;
|
|
19
|
+
/** Click-target selector -> the file that click produces, as `filename:contents`. */
|
|
20
|
+
readonly downloads?: Readonly<Record<string, string>> | undefined;
|
|
21
|
+
/** ISO 8601. Absent means "no age", and `fixtureBrowser({ maxAge })` cannot judge it. */
|
|
22
|
+
readonly recordedAt?: string | undefined;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
export const pageRecordingSchema: StandardSchemaV1<unknown, PageRecording> = t.object({
|
|
26
|
+
url: t.string,
|
|
27
|
+
html: t.string,
|
|
28
|
+
evaluate: t.optional(t.record(t.string)),
|
|
29
|
+
frames: t.optional(t.record(t.string)),
|
|
30
|
+
downloads: t.optional(t.record(t.string)),
|
|
31
|
+
recordedAt: t.optional(t.string),
|
|
32
|
+
}) as unknown as StandardSchemaV1<unknown, PageRecording>;
|
|
33
|
+
|
|
34
|
+
/** On-disk JSON is `unknown` and is PARSED, never cast — a fixture is somebody's edited file. */
|
|
35
|
+
export const parseRecording = (raw: unknown): PageRecording => parse(pageRecordingSchema, raw);
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* The HTTP leg's half of the SAME fixture directory. One format covers both transports, because a
|
|
39
|
+
* hybrid scrape — browser login, session handoff, HTTP bulk fetch — is only testable if the whole
|
|
40
|
+
* run replays from one place. Split them across two mechanisms and the hybrid path, which is where
|
|
41
|
+
* the real code lives, becomes the untestable path.
|
|
42
|
+
*/
|
|
43
|
+
export interface HttpRecording {
|
|
44
|
+
readonly url: string;
|
|
45
|
+
readonly method: string;
|
|
46
|
+
readonly status: number;
|
|
47
|
+
readonly body: string;
|
|
48
|
+
readonly headers?: Readonly<Record<string, string>> | undefined;
|
|
49
|
+
readonly recordedAt?: string | undefined;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
export const httpRecordingSchema: StandardSchemaV1<unknown, HttpRecording> = t.object({
|
|
53
|
+
url: t.string,
|
|
54
|
+
method: t.string,
|
|
55
|
+
status: t.number,
|
|
56
|
+
body: t.string,
|
|
57
|
+
headers: t.optional(t.record(t.string)),
|
|
58
|
+
recordedAt: t.optional(t.string),
|
|
59
|
+
}) as unknown as StandardSchemaV1<unknown, HttpRecording>;
|
|
60
|
+
|
|
61
|
+
export const parseHttpRecording = (raw: unknown): HttpRecording => parse(httpRecordingSchema, raw);
|
|
62
|
+
|
|
63
|
+
/** `report.csv:a,b,c` -> the two halves. A value with no colon is all contents, no name. */
|
|
64
|
+
export function splitDownload(value: string): { filename: string; contents: string } {
|
|
65
|
+
const colon = value.indexOf(':');
|
|
66
|
+
if (colon === -1) return { filename: 'download', contents: value };
|
|
67
|
+
return { filename: value.slice(0, colon), contents: value.slice(colon + 1) };
|
|
68
|
+
}
|
package/src/recover.ts
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
// The recovery seam: what happens when a selector has moved and the run failed on it.
|
|
2
|
+
//
|
|
3
|
+
// Two shapes. A FUNCTION is complete and shipped — an app writes its own fallback, gets the page
|
|
4
|
+
// and the failure, and answers whether the attempt should be retried. `'agent'` is the seam for
|
|
5
|
+
// letting a model re-derive the selector from the page, and it is an honest
|
|
6
|
+
// `X_NOT_IMPLEMENTED` today, in the shape `packages/jobs/src/driver-redis.ts` uses: correct types
|
|
7
|
+
// so an app can be written against it, one labelled throw, no silent no-op.
|
|
8
|
+
|
|
9
|
+
import { recoverRefused, scrapeNotImplemented } from './error-throws';
|
|
10
|
+
import type { ScrapePage } from './page';
|
|
11
|
+
|
|
12
|
+
export interface RecoveryAttempt {
|
|
13
|
+
readonly scrape: string;
|
|
14
|
+
readonly page: ScrapePage;
|
|
15
|
+
/** The failure, as thrown. `unknown` — it is whatever the body raised. */
|
|
16
|
+
readonly failure: unknown;
|
|
17
|
+
readonly attempt: number;
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* `true` retries the body once more in the same session; `false` re-raises. A hook that returns
|
|
22
|
+
* `false` is not an error — declining is a decision. `X_SCRAPE_RECOVER_REFUSED` is for the hook
|
|
23
|
+
* that cannot even judge, and it carries the hook's own reason.
|
|
24
|
+
*/
|
|
25
|
+
export type RecoveryHook = (attempt: RecoveryAttempt) => Promise<boolean> | boolean;
|
|
26
|
+
|
|
27
|
+
export type Recovery = RecoveryHook | 'agent';
|
|
28
|
+
|
|
29
|
+
/** What the agent recovery will need when it lands. Declared now so the seam is typed, not free. */
|
|
30
|
+
export interface AgentRecovery {
|
|
31
|
+
readonly kind: 'agent';
|
|
32
|
+
/** The model call. Deliberately not typed against `@ultimat3/ai` yet — see the throw below. */
|
|
33
|
+
readonly instruction?: string | undefined;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
const AGENT_FIX =
|
|
37
|
+
"replace recover: 'agent' with a function — recover: ({ page }) => page.waitFor('<the moved selector>').then(() => true) — which is complete today";
|
|
38
|
+
|
|
39
|
+
export async function runRecovery(recovery: Recovery, attempt: RecoveryAttempt): Promise<boolean> {
|
|
40
|
+
if (recovery === 'agent') {
|
|
41
|
+
// The agent path lands with `@ultimat3/ai`'s `llm()` behind it: page HTML in, a selector out,
|
|
42
|
+
// re-run once. Until it does, this throws rather than answering `false` — a recovery that
|
|
43
|
+
// silently declines is indistinguishable from one that was never configured.
|
|
44
|
+
throw scrapeNotImplemented("recover: 'agent'", AGENT_FIX);
|
|
45
|
+
}
|
|
46
|
+
const verdict: unknown = await recovery(attempt);
|
|
47
|
+
if (typeof verdict !== 'boolean') {
|
|
48
|
+
throw recoverRefused(attempt.scrape, 'the hook answered something other than true or false');
|
|
49
|
+
}
|
|
50
|
+
return verdict;
|
|
51
|
+
}
|
package/src/rings.ts
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
// Bounded console and network history. BOUNDED is the whole point: a scrape of ten thousand
|
|
2
|
+
// pages that kept every console line and every request holds the run's entire browsing history in
|
|
3
|
+
// the worker's heap, and the incident is an OOM two hours in rather than a scraper bug.
|
|
4
|
+
|
|
5
|
+
export interface ConsoleLine {
|
|
6
|
+
readonly level: 'log' | 'info' | 'warn' | 'error' | 'debug';
|
|
7
|
+
readonly text: string;
|
|
8
|
+
readonly at: number;
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
export interface NetworkEntry {
|
|
12
|
+
readonly method: string;
|
|
13
|
+
readonly url: string;
|
|
14
|
+
/** Absent while the request is still in flight or was aborted by interception. */
|
|
15
|
+
readonly status?: number | undefined;
|
|
16
|
+
readonly resourceType: ResourceType;
|
|
17
|
+
readonly at: number;
|
|
18
|
+
/** Set when interception refused it — `blocked` for `block:`, `host` for `allowHosts`. */
|
|
19
|
+
readonly refused?: 'blocked' | 'host' | 'robots' | undefined;
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
export const RESOURCE_TYPES = [
|
|
23
|
+
'document',
|
|
24
|
+
'stylesheet',
|
|
25
|
+
'image',
|
|
26
|
+
'media',
|
|
27
|
+
'font',
|
|
28
|
+
'script',
|
|
29
|
+
'xhr',
|
|
30
|
+
'fetch',
|
|
31
|
+
'websocket',
|
|
32
|
+
'other',
|
|
33
|
+
] as const;
|
|
34
|
+
|
|
35
|
+
export type ResourceType = (typeof RESOURCE_TYPES)[number];
|
|
36
|
+
|
|
37
|
+
export const DEFAULT_RING_CAPACITY = 200;
|
|
38
|
+
|
|
39
|
+
export interface Ring<T> {
|
|
40
|
+
readonly capacity: number;
|
|
41
|
+
push(entry: T): void;
|
|
42
|
+
/** Oldest first. A copy — a caller holding the ring's own array would see it mutate. */
|
|
43
|
+
entries(): readonly T[];
|
|
44
|
+
/** How many were dropped to keep the bound. Non-zero is the honest "you are not seeing it all". */
|
|
45
|
+
readonly dropped: number;
|
|
46
|
+
clear(): void;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
export type ConsoleRing = Ring<ConsoleLine>;
|
|
50
|
+
export type NetworkRing = Ring<NetworkEntry>;
|
|
51
|
+
|
|
52
|
+
export function createRing<T>(capacity: number = DEFAULT_RING_CAPACITY): Ring<T> {
|
|
53
|
+
const items: T[] = [];
|
|
54
|
+
let dropped = 0;
|
|
55
|
+
return {
|
|
56
|
+
capacity,
|
|
57
|
+
push(entry: T): void {
|
|
58
|
+
items.push(entry);
|
|
59
|
+
while (items.length > capacity) {
|
|
60
|
+
items.shift();
|
|
61
|
+
dropped += 1;
|
|
62
|
+
}
|
|
63
|
+
},
|
|
64
|
+
entries: () => [...items],
|
|
65
|
+
get dropped(): number {
|
|
66
|
+
return dropped;
|
|
67
|
+
},
|
|
68
|
+
clear(): void {
|
|
69
|
+
items.length = 0;
|
|
70
|
+
dropped = 0;
|
|
71
|
+
},
|
|
72
|
+
};
|
|
73
|
+
}
|
package/src/robots.ts
ADDED
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
// robots.txt, obeyed by DEFAULT, and ignorable only with a written reason.
|
|
2
|
+
//
|
|
3
|
+
// Ultimate's users are not all scraping their own accounts. `robots: 'obey'` being the default
|
|
4
|
+
// makes the polite thing the thing that happens when nobody decides; `robots: { ignore: reason }`
|
|
5
|
+
// makes the impolite thing a sentence somebody has to write, in the diff, next to their name.
|
|
6
|
+
// There is no boolean, because `robots: false` is a decision with no author.
|
|
7
|
+
|
|
8
|
+
import { robotsDisallowed } from './error-throws';
|
|
9
|
+
|
|
10
|
+
export type RobotsPolicy = 'obey' | { readonly ignore: string };
|
|
11
|
+
|
|
12
|
+
export const DEFAULT_ROBOTS_AGENT = 'ultimate-scraper';
|
|
13
|
+
|
|
14
|
+
export interface RobotsRule {
|
|
15
|
+
readonly allow: boolean;
|
|
16
|
+
readonly path: string;
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
export interface RobotsRules {
|
|
20
|
+
readonly rules: readonly RobotsRule[];
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* The `User-agent` groups that apply to `agent`: its own, plus `*` when the file names no group
|
|
25
|
+
* for it. A file that addresses this agent specifically REPLACES the wildcard group, which is
|
|
26
|
+
* what the standard says and what every operator writing a targeted rule expects.
|
|
27
|
+
*/
|
|
28
|
+
export function parseRobots(text: string, agent: string): RobotsRules {
|
|
29
|
+
const wanted = agent.toLowerCase();
|
|
30
|
+
const groups = new Map<string, RobotsRule[]>();
|
|
31
|
+
let current: string[] = [];
|
|
32
|
+
let inGroup = false;
|
|
33
|
+
for (const raw of text.split('\n')) {
|
|
34
|
+
const line = raw.split('#')[0]?.trim() ?? '';
|
|
35
|
+
if (line === '') continue;
|
|
36
|
+
const colon = line.indexOf(':');
|
|
37
|
+
if (colon === -1) continue;
|
|
38
|
+
const field = line.slice(0, colon).trim().toLowerCase();
|
|
39
|
+
const value = line.slice(colon + 1).trim();
|
|
40
|
+
if (field === 'user-agent') {
|
|
41
|
+
if (inGroup) current = [];
|
|
42
|
+
inGroup = false;
|
|
43
|
+
current.push(value.toLowerCase());
|
|
44
|
+
continue;
|
|
45
|
+
}
|
|
46
|
+
if (field !== 'allow' && field !== 'disallow') continue;
|
|
47
|
+
inGroup = true;
|
|
48
|
+
for (const name of current) {
|
|
49
|
+
const rules = groups.get(name) ?? [];
|
|
50
|
+
// An empty `Disallow:` is the documented spelling of "allow everything", not a rule about
|
|
51
|
+
// the empty path — read as a path it would match every URL and refuse the whole site.
|
|
52
|
+
if (value !== '' || field === 'allow') rules.push({ allow: field === 'allow', path: value });
|
|
53
|
+
groups.set(name, rules);
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
return { rules: groups.get(wanted) ?? groups.get('*') ?? [] };
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
const escaped = (literal: string): string => literal.replaceAll(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
60
|
+
|
|
61
|
+
/** `/private/*.pdf$` — the two wildcards robots.txt defines, and no others. */
|
|
62
|
+
const patternMatches = (pattern: string, path: string): boolean => {
|
|
63
|
+
const anchored = pattern.endsWith('$');
|
|
64
|
+
const body = anchored ? pattern.slice(0, -1) : pattern;
|
|
65
|
+
const source = body.split('*').map(escaped).join('.*');
|
|
66
|
+
return new RegExp(`^${source}${anchored ? '$' : ''}`).test(path);
|
|
67
|
+
};
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* Longest match wins, and a tie goes to `Allow` — the rule every major crawler implements. A
|
|
71
|
+
* naive "first disallow wins" reading refuses paths an operator explicitly re-allowed.
|
|
72
|
+
*/
|
|
73
|
+
export function robotsAllows(rules: RobotsRules, path: string): boolean {
|
|
74
|
+
let verdict = true;
|
|
75
|
+
let best = -1;
|
|
76
|
+
for (const rule of rules.rules) {
|
|
77
|
+
if (!patternMatches(rule.path, path)) continue;
|
|
78
|
+
const weight = rule.path.length;
|
|
79
|
+
if (weight > best || (weight === best && rule.allow)) {
|
|
80
|
+
best = weight;
|
|
81
|
+
verdict = rule.allow;
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
return verdict;
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
export interface RobotsGate {
|
|
88
|
+
/** Throws `X_SCRAPE_ROBOTS_DISALLOWED`, or returns. Called before every navigation. */
|
|
89
|
+
assertAllowed(url: string): Promise<void>;
|
|
90
|
+
/** The reason an `ignore` gate carries, for the run's own record. `undefined` when obeying. */
|
|
91
|
+
readonly ignoredBecause: string | undefined;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
export type RobotsFetch = (robotsUrl: string) => Promise<string | undefined>;
|
|
95
|
+
|
|
96
|
+
const fetchRobots: RobotsFetch = async (robotsUrl) => {
|
|
97
|
+
const response = await fetch(robotsUrl);
|
|
98
|
+
return response.ok ? await response.text() : undefined;
|
|
99
|
+
};
|
|
100
|
+
|
|
101
|
+
export interface RobotsGateInit {
|
|
102
|
+
readonly policy: RobotsPolicy;
|
|
103
|
+
readonly agent?: string | undefined;
|
|
104
|
+
readonly fetchText?: RobotsFetch | undefined;
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/**
|
|
108
|
+
* One fetch of `/robots.txt` per ORIGIN per run, cached. A gate that re-fetched per navigation
|
|
109
|
+
* would triple the request count of every scrape it protects.
|
|
110
|
+
*
|
|
111
|
+
* An origin whose robots.txt cannot be read is ALLOWED. That is the standard's own answer — a
|
|
112
|
+
* missing file means no restrictions — and the alternative fails every run behind a flaky CDN.
|
|
113
|
+
*/
|
|
114
|
+
export function createRobotsGate(init: RobotsGateInit): RobotsGate {
|
|
115
|
+
if (init.policy !== 'obey') {
|
|
116
|
+
return { assertAllowed: () => Promise.resolve(), ignoredBecause: init.policy.ignore };
|
|
117
|
+
}
|
|
118
|
+
const agent = init.agent ?? DEFAULT_ROBOTS_AGENT;
|
|
119
|
+
const read = init.fetchText ?? fetchRobots;
|
|
120
|
+
const cache = new Map<string, Promise<RobotsRules>>();
|
|
121
|
+
return {
|
|
122
|
+
ignoredBecause: undefined,
|
|
123
|
+
async assertAllowed(url: string): Promise<void> {
|
|
124
|
+
let parsed: URL;
|
|
125
|
+
try {
|
|
126
|
+
parsed = new URL(url);
|
|
127
|
+
} catch {
|
|
128
|
+
return;
|
|
129
|
+
}
|
|
130
|
+
if (parsed.protocol !== 'http:' && parsed.protocol !== 'https:') return;
|
|
131
|
+
const origin = parsed.origin;
|
|
132
|
+
let rules = cache.get(origin);
|
|
133
|
+
if (rules === undefined) {
|
|
134
|
+
rules = read(`${origin}/robots.txt`)
|
|
135
|
+
.then((text) => (text === undefined ? { rules: [] } : parseRobots(text, agent)))
|
|
136
|
+
.catch(() => ({ rules: [] }));
|
|
137
|
+
cache.set(origin, rules);
|
|
138
|
+
}
|
|
139
|
+
if (!robotsAllows(await rules, `${parsed.pathname}${parsed.search}`)) {
|
|
140
|
+
throw robotsDisallowed(url, agent);
|
|
141
|
+
}
|
|
142
|
+
},
|
|
143
|
+
};
|
|
144
|
+
}
|