@ultimat3/scraping 2.0.0 → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +5 -5
- package/src/artifacts.ts +18 -9
- package/src/cdp-target.ts +36 -9
- package/src/driver-cdp.ts +21 -2
- package/src/driver.ts +9 -0
- package/src/error-throws.ts +28 -0
- package/src/errors.ts +11 -0
- package/src/html-target.ts +24 -9
- package/src/http.ts +58 -7
- package/src/index.ts +14 -2
- package/src/page-over-target.ts +8 -5
- package/src/page.ts +1 -1
- package/src/robots-fetch.ts +82 -0
- package/src/robots.ts +44 -13
- package/src/scrape-run.ts +48 -4
- package/src/scrape.ts +7 -0
- package/src/secrets.ts +18 -4
- package/src/session-state.ts +30 -6
- package/src/target.ts +19 -2
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ultimat3/scraping",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "4.0.0",
|
|
4
4
|
"description": "Browser automation as a job: scrape() returns a JobHandle",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"type": "module",
|
|
@@ -30,9 +30,9 @@
|
|
|
30
30
|
"test": "bun test"
|
|
31
31
|
},
|
|
32
32
|
"dependencies": {
|
|
33
|
-
"@ultimat3/core": "
|
|
34
|
-
"@ultimat3/jobs": "
|
|
35
|
-
"@ultimat3/schema": "
|
|
36
|
-
"@ultimat3/storage": "
|
|
33
|
+
"@ultimat3/core": "4.0.0",
|
|
34
|
+
"@ultimat3/jobs": "4.0.0",
|
|
35
|
+
"@ultimat3/schema": "4.0.0",
|
|
36
|
+
"@ultimat3/storage": "4.0.0"
|
|
37
37
|
}
|
|
38
38
|
}
|
package/src/artifacts.ts
CHANGED
|
@@ -30,17 +30,26 @@ export interface ArtifactWriterInit {
|
|
|
30
30
|
|
|
31
31
|
export const DEFAULT_ARTIFACT_PREFIX = 'scrape';
|
|
32
32
|
|
|
33
|
-
const
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
33
|
+
export const DEFAULT_CONTENT_TYPE = 'application/octet-stream';
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* A `Map`, not an object literal — the same choice `failures.ts` makes, for the same reason. The
|
|
37
|
+
* extension comes off a caller-supplied filename, and on a download that filename came off the
|
|
38
|
+
* site's own `Content-Disposition`: indexing a plain object with it answered a FUNCTION for
|
|
39
|
+
* `report.constructor` and an object for `report.__proto__`, where the type says `string`, and
|
|
40
|
+
* that non-string went on to be `ref.contentType` and an S3 header.
|
|
41
|
+
*/
|
|
42
|
+
const CONTENT_TYPES: ReadonlyMap<string, string> = new Map([
|
|
43
|
+
['html', 'text/html; charset=utf-8'],
|
|
44
|
+
['png', 'image/png'],
|
|
45
|
+
['pdf', 'application/pdf'],
|
|
46
|
+
['json', 'application/json'],
|
|
47
|
+
['csv', 'text/csv'],
|
|
48
|
+
['txt', 'text/plain; charset=utf-8'],
|
|
49
|
+
]);
|
|
41
50
|
|
|
42
51
|
export const contentTypeFor = (name: string): string =>
|
|
43
|
-
CONTENT_TYPES
|
|
52
|
+
CONTENT_TYPES.get(name.split('.').pop()?.toLowerCase() ?? '') ?? DEFAULT_CONTENT_TYPE;
|
|
44
53
|
|
|
45
54
|
/**
|
|
46
55
|
* A writer with no storage driver is a NO-OP that still answers a key — never a throw. An app
|
package/src/cdp-target.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
// `ScrapeTarget` over a real browser, through the structural CDP port. Everything driver-specific
|
|
2
2
|
// in this package lives here and in `driver-cdp.ts`; the vocabulary above it does not change.
|
|
3
3
|
|
|
4
|
+
import { isUltimateError } from '@ultimat3/core';
|
|
4
5
|
import type { StandardSchemaV1 } from '@ultimat3/schema';
|
|
5
6
|
import { parse, t } from '@ultimat3/schema';
|
|
6
7
|
import type { CdpBrowserLike, CdpFrameLike, CdpPageLike, CdpRequestLike } from './cdp-port';
|
|
@@ -87,6 +88,26 @@ const readStringFrom = (owner: unknown, key: string): string | undefined => {
|
|
|
87
88
|
return typeof answer === 'string' ? answer : undefined;
|
|
88
89
|
};
|
|
89
90
|
|
|
91
|
+
/**
|
|
92
|
+
* The one failure `guard()` must NOT re-label, and the line is drawn at exactly one code.
|
|
93
|
+
*
|
|
94
|
+
* `X_NOT_IMPLEMENTED` is the only code that says "this build does not have the feature" — a fact
|
|
95
|
+
* about the launcher's own shape, never about the connection. A browser cannot produce it; only
|
|
96
|
+
* this file's own `scrapeNotImplemented()` can, from inside a guarded closure. Re-labelled as
|
|
97
|
+
* `X_SCRAPE_BROWSER_UNREACHABLE` (registered `retryable` in `errors.ts`) it spends every attempt
|
|
98
|
+
* in the scrape's retry policy on a method that is still missing on attempt five, and tells the
|
|
99
|
+
* operator the browser went away while the browser is answering fine.
|
|
100
|
+
*
|
|
101
|
+
* Every OTHER coded error stays wrapped, deliberately. `thrown instanceof UltimateError` is the
|
|
102
|
+
* naive version of this check and it is wrong: an `X_SCRAPE_TIMEOUT` raised while the socket was
|
|
103
|
+
* already dead would then arrive unwrapped, and "the browser went away mid-run" is the frame that
|
|
104
|
+
* makes a disconnect legible — which is the whole reason this wrapper exists.
|
|
105
|
+
*
|
|
106
|
+
* `isUltimateError`, not `instanceof`: the brand survives a duplicated module instance.
|
|
107
|
+
*/
|
|
108
|
+
const isStructuralRefusal = (thrown: unknown): boolean =>
|
|
109
|
+
isUltimateError(thrown) && thrown.code === 'X_NOT_IMPLEMENTED';
|
|
110
|
+
|
|
90
111
|
export interface CdpTargetInit {
|
|
91
112
|
readonly page: CdpPageLike;
|
|
92
113
|
readonly browser: CdpBrowserLike;
|
|
@@ -185,6 +206,7 @@ export async function cdpTarget(init: CdpTargetInit): Promise<ScrapeTarget> {
|
|
|
185
206
|
return await run();
|
|
186
207
|
} catch (thrown) {
|
|
187
208
|
live();
|
|
209
|
+
if (isStructuralRefusal(thrown)) throw thrown;
|
|
188
210
|
throw browserUnreachable(`${CDP_DRIVER} ${what}`, thrown);
|
|
189
211
|
}
|
|
190
212
|
};
|
|
@@ -256,15 +278,20 @@ export async function cdpTarget(init: CdpTargetInit): Promise<ScrapeTarget> {
|
|
|
256
278
|
}
|
|
257
279
|
return parse(cookieSchema, await source.cookies());
|
|
258
280
|
}),
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
281
|
+
// Honest stub, in the shape `packages/jobs/src/driver-redis.ts` uses. A real one needs
|
|
282
|
+
// `Browser.setDownloadBehavior` over a raw CDP session plus a directory watch, and a
|
|
283
|
+
// half-written version that silently returned empty bytes is worse than this line.
|
|
284
|
+
//
|
|
285
|
+
// REJECTS rather than throws: the method is typed `Promise<ScrapeDownloadFile>` and every
|
|
286
|
+
// caller of a promise-typed method handles its failure with `.catch()` or an `await` inside a
|
|
287
|
+
// `try` — a synchronous `throw` jumps over the first of those entirely.
|
|
288
|
+
download: (_options): Promise<ScrapeDownloadFile> =>
|
|
289
|
+
Promise.reject(
|
|
290
|
+
scrapeNotImplemented(
|
|
291
|
+
'download() on the puppeteer driver',
|
|
292
|
+
'fetch the file inside the page — page.evaluate("fetch(url).then(r => r.text())") — or run this scrape on fixtureBrowser(), whose download() is complete',
|
|
293
|
+
),
|
|
294
|
+
),
|
|
268
295
|
frames: () =>
|
|
269
296
|
guard('frames', () =>
|
|
270
297
|
Promise.resolve(
|
package/src/driver-cdp.ts
CHANGED
|
@@ -67,6 +67,19 @@ async function opened(browser: CdpBrowserLike, init: SessionInit): Promise<Scrap
|
|
|
67
67
|
}
|
|
68
68
|
}
|
|
69
69
|
|
|
70
|
+
/**
|
|
71
|
+
* The run's cancellation and the watchdog's, as ONE signal handed to every wait.
|
|
72
|
+
*
|
|
73
|
+
* The guard's abort half had no reader: both legs below were passed `init.signal`, so the
|
|
74
|
+
* watchdog's only production effect was `kill()` — and `Browser.process()` answers `null` for a
|
|
75
|
+
* browser obtained through `connect()`, which is `remoteBrowser()`, this file's primary path. A
|
|
76
|
+
* wedge therefore killed nothing and aborted a signal nobody composed, and the blocked await on
|
|
77
|
+
* the CDP socket stayed blocked past `ctx.signal` and past the watchdog. Verbatim incident #1 in
|
|
78
|
+
* `watchdog.ts`, unfixed for the attach path until the composition below.
|
|
79
|
+
*/
|
|
80
|
+
const withWedgeSignal = (run: AbortSignal | undefined, guard: AbortSignal): AbortSignal =>
|
|
81
|
+
run === undefined ? guard : AbortSignal.any([run, guard]);
|
|
82
|
+
|
|
70
83
|
async function sessionOver(
|
|
71
84
|
browser: CdpBrowserLike,
|
|
72
85
|
init: SessionInit,
|
|
@@ -99,15 +112,21 @@ async function sessionOver(
|
|
|
99
112
|
guard.touch();
|
|
100
113
|
init.onActivity?.();
|
|
101
114
|
};
|
|
115
|
+
const signal = withWedgeSignal(init.signal, guard.signal);
|
|
102
116
|
return {
|
|
103
117
|
driver: CDP_DRIVER,
|
|
118
|
+
// The exit the browser was launched or attached with, handed back so the run's robots read
|
|
119
|
+
// presents the SAME client identity to the origin the page loads from. `options.proxy` and
|
|
120
|
+
// not `init.proxy`: this is what the launch args and the HTTP leg below actually carry, and a
|
|
121
|
+
// reported exit that nothing dialled would be worse than none.
|
|
122
|
+
...(options.proxy === undefined ? {} : { proxy: options.proxy }),
|
|
104
123
|
page: pageOverTarget(target, {
|
|
105
124
|
clock: init.clock,
|
|
106
125
|
allowHosts: init.rules.allowHosts,
|
|
107
126
|
defaultTimeoutMs: init.timeoutMs,
|
|
108
127
|
secrets: init.secrets,
|
|
109
128
|
robots: init.robots,
|
|
110
|
-
signal
|
|
129
|
+
signal,
|
|
111
130
|
onActivity,
|
|
112
131
|
pace: init.pace,
|
|
113
132
|
}),
|
|
@@ -121,7 +140,7 @@ async function sessionOver(
|
|
|
121
140
|
session: () => target.session(),
|
|
122
141
|
robots: init.robots,
|
|
123
142
|
pace: init.pace,
|
|
124
|
-
signal
|
|
143
|
+
signal,
|
|
125
144
|
onActivity,
|
|
126
145
|
proxy: options.proxy,
|
|
127
146
|
}),
|
package/src/driver.ts
CHANGED
|
@@ -42,6 +42,15 @@ export interface SessionInit {
|
|
|
42
42
|
|
|
43
43
|
export interface ScrapeSession {
|
|
44
44
|
readonly driver: string;
|
|
45
|
+
/**
|
|
46
|
+
* The exit this session ACTUALLY dialled — `undefined` for a direct one. Reported because the
|
|
47
|
+
* proxy is resolved inside `open()`, after the robots gate handed to it was built: without this
|
|
48
|
+
* the default `/robots.txt` read leaves from the worker's IP while every page load leaves
|
|
49
|
+
* through the proxy, which is a second client identity presented to the same origin — and an
|
|
50
|
+
* origin reachable only through the proxy then reads as "no robots.txt", which the gate treats
|
|
51
|
+
* as allow-everything. A driver that omits it is asked for its rules directly, as before.
|
|
52
|
+
*/
|
|
53
|
+
readonly proxy?: string | undefined;
|
|
45
54
|
readonly page: ScrapePage;
|
|
46
55
|
/**
|
|
47
56
|
* The second transport, bound to the SAME session as the page: the browser's cookies, headers
|
package/src/error-throws.ts
CHANGED
|
@@ -75,6 +75,21 @@ export const notActionable = (selector: string, reason: string, waitedMs: number
|
|
|
75
75
|
meta: { selector, reason, waitedMs },
|
|
76
76
|
});
|
|
77
77
|
|
|
78
|
+
/**
|
|
79
|
+
* Declared, and structurally unable to fire. `maxDrop` is a fraction of a TRAILING MEDIAN, and the
|
|
80
|
+
* only source of one is a `history:` store — with none, `guardYield` reads an empty array and the
|
|
81
|
+
* `MIN_BASELINE_RUNS` gate is true forever, so the alarm never raises `X_SCRAPE_YIELD_COLLAPSED`
|
|
82
|
+
* on any run. Two halves that must be set together, where setting one alone did nothing and
|
|
83
|
+
* nothing said so.
|
|
84
|
+
*/
|
|
85
|
+
export const yieldHistoryMissing = (scrapeName: string): ScrapeError =>
|
|
86
|
+
new ScrapeError({
|
|
87
|
+
code: 'X_SCRAPE_YIELD_HISTORY_MISSING',
|
|
88
|
+
cause: `scrape "${scrapeName}" declares expect.maxDrop with no history: store, so there is no baseline to measure a drop against and the alarm can never fire`,
|
|
89
|
+
fix: `add history: memoryYieldHistory() to scrape("${scrapeName}") for a test, or your own YieldHistory in production — or drop expect.maxDrop and keep expect.minRows, which needs no baseline`,
|
|
90
|
+
meta: { scrape: scrapeName },
|
|
91
|
+
});
|
|
92
|
+
|
|
78
93
|
export const scrapeTimeout = (what: string, ms: number): ScrapeError =>
|
|
79
94
|
new ScrapeError({
|
|
80
95
|
code: 'X_SCRAPE_TIMEOUT',
|
|
@@ -203,6 +218,19 @@ export const httpFailed = (url: string, status: number, body: string): ScrapeErr
|
|
|
203
218
|
retry: status >= 400 && status < 500 && status !== 429 ? 'terminal' : 'retryable',
|
|
204
219
|
});
|
|
205
220
|
|
|
221
|
+
/**
|
|
222
|
+
* The body outgrew its cap. `AbortSignal.timeout` bounds TIME, not bytes — a 30s stream at 50MB/s
|
|
223
|
+
* is a 1.5GB allocation — so a hostile or merely paginated endpoint could OOM-kill the worker
|
|
224
|
+
* mid-run, taking every other job on it with it.
|
|
225
|
+
*/
|
|
226
|
+
export const bodyTooLarge = (url: string, readBytes: number, maxBytes: number): ScrapeError =>
|
|
227
|
+
new ScrapeError({
|
|
228
|
+
code: 'X_SCRAPE_BODY_TOO_LARGE',
|
|
229
|
+
cause: `${url} sent more than ${String(maxBytes)} bytes (${String(readBytes)} read before the read was cancelled)`,
|
|
230
|
+
fix: `raise the cap on this one call — http.request(url, { maxBytes: ${String(maxBytes * 2)} }) — or ask the endpoint for a page instead of the whole collection`,
|
|
231
|
+
meta: { url, readBytes, maxBytes },
|
|
232
|
+
});
|
|
233
|
+
|
|
206
234
|
/**
|
|
207
235
|
* TERMINAL, and the retry table cannot be talked out of it. A site that locks an account after
|
|
208
236
|
* three wrong attempts turns a retrying framework into the thing that destroys the user's
|
package/src/errors.ts
CHANGED
|
@@ -25,6 +25,7 @@ export const SCRAPE_OWNED_ERROR_CODES = [
|
|
|
25
25
|
'X_SCRAPE_PAGE_CRASHED',
|
|
26
26
|
'X_SCRAPE_OUTPUT_INVALID',
|
|
27
27
|
'X_SCRAPE_YIELD_COLLAPSED',
|
|
28
|
+
'X_SCRAPE_YIELD_HISTORY_MISSING',
|
|
28
29
|
'X_SCRAPE_DOWNLOAD_TIMEOUT',
|
|
29
30
|
'X_SCRAPE_ROBOTS_DISALLOWED',
|
|
30
31
|
'X_SCRAPE_FIXTURE_MISSING',
|
|
@@ -33,6 +34,7 @@ export const SCRAPE_OWNED_ERROR_CODES = [
|
|
|
33
34
|
'X_SCRAPE_RECOVER_REFUSED',
|
|
34
35
|
'X_SCRAPE_SECRET_EXPOSED',
|
|
35
36
|
'X_SCRAPE_HTTP_FAILED',
|
|
37
|
+
'X_SCRAPE_BODY_TOO_LARGE',
|
|
36
38
|
'X_SCRAPE_AUTH_FAILED',
|
|
37
39
|
'X_SCRAPE_SESSION_EXPIRED',
|
|
38
40
|
'X_SCRAPE_PROMPT_UNANSWERED',
|
|
@@ -69,6 +71,8 @@ export const SCRAPE_ERROR_TITLES: Readonly<Record<ScrapeOwnedErrorCode, string>>
|
|
|
69
71
|
X_SCRAPE_PAGE_CRASHED: 'the renderer process died',
|
|
70
72
|
X_SCRAPE_OUTPUT_INVALID: 'the extracted rows do not match the extract schema',
|
|
71
73
|
X_SCRAPE_YIELD_COLLAPSED: 'the run succeeded and returned far too little',
|
|
74
|
+
X_SCRAPE_YIELD_HISTORY_MISSING:
|
|
75
|
+
'a maxDrop is declared and no history store can supply its baseline',
|
|
72
76
|
X_SCRAPE_DOWNLOAD_TIMEOUT: 'the download never landed',
|
|
73
77
|
X_SCRAPE_ROBOTS_DISALLOWED: 'robots.txt disallows this path',
|
|
74
78
|
X_SCRAPE_FIXTURE_MISSING: 'the fixture driver has no recording for this request',
|
|
@@ -77,6 +81,7 @@ export const SCRAPE_ERROR_TITLES: Readonly<Record<ScrapeOwnedErrorCode, string>>
|
|
|
77
81
|
X_SCRAPE_RECOVER_REFUSED: 'the recovery hook declined to recover this failure',
|
|
78
82
|
X_SCRAPE_SECRET_EXPOSED: 'an artifact would have carried a secret this run typed',
|
|
79
83
|
X_SCRAPE_HTTP_FAILED: 'the site answered the HTTP leg with a non-2xx status',
|
|
84
|
+
X_SCRAPE_BODY_TOO_LARGE: 'the HTTP response body passed its byte cap',
|
|
80
85
|
X_SCRAPE_AUTH_FAILED: 'the credentials were rejected',
|
|
81
86
|
X_SCRAPE_SESSION_EXPIRED: 'the restored session is no longer valid and nothing can renew it',
|
|
82
87
|
X_SCRAPE_PROMPT_UNANSWERED: 'a login step asked for a code and nothing answered',
|
|
@@ -126,7 +131,13 @@ export const SCRAPE_ERROR_RETRY = {
|
|
|
126
131
|
// half-submitted form is the incident, not the recovery.
|
|
127
132
|
X_SCRAPE_PAGE_CRASHED: 'terminal',
|
|
128
133
|
X_SCRAPE_OUTPUT_INVALID: 'terminal',
|
|
134
|
+
// A response size is a property of the endpoint, not of the moment: attempt 2 buffers the same
|
|
135
|
+
// gigabyte and dies the same way. The fix is a number on the request, so a human decides it.
|
|
136
|
+
X_SCRAPE_BODY_TOO_LARGE: 'terminal',
|
|
129
137
|
X_SCRAPE_YIELD_COLLAPSED: 'terminal',
|
|
138
|
+
// A declaration error, raised by `scrape()` before any attempt exists — there is no run to
|
|
139
|
+
// retry, and the same definition would refuse identically forever.
|
|
140
|
+
X_SCRAPE_YIELD_HISTORY_MISSING: 'terminal',
|
|
130
141
|
X_SCRAPE_ROBOTS_DISALLOWED: 'terminal',
|
|
131
142
|
X_SCRAPE_FIXTURE_MISSING: 'terminal',
|
|
132
143
|
X_SCRAPE_FIXTURE_STALE: 'terminal',
|
package/src/html-target.ts
CHANGED
|
@@ -49,6 +49,18 @@ export interface HtmlTargetInit {
|
|
|
49
49
|
|
|
50
50
|
const EMPTY: PageRecording = { url: 'about:blank', html: '' };
|
|
51
51
|
|
|
52
|
+
/**
|
|
53
|
+
* A recorded map, read by a key that came out of the RECORDING's own markup — a selector, an
|
|
54
|
+
* expression, an `<iframe name>`. Plain indexing walks the prototype chain, so `name="constructor"`
|
|
55
|
+
* resolved to `Object` rather than to `undefined` and `fixtureMissing` never threw; the run then
|
|
56
|
+
* carried a function where an HTML string belongs. Offline driver only, and it is still worth
|
|
57
|
+
* closing: the confusing test failure it produces costs more to read than this line does.
|
|
58
|
+
*/
|
|
59
|
+
const recorded = (
|
|
60
|
+
map: Readonly<Record<string, string>> | undefined,
|
|
61
|
+
key: string,
|
|
62
|
+
): string | undefined => (map !== undefined && Object.hasOwn(map, key) ? map[key] : undefined);
|
|
63
|
+
|
|
52
64
|
/** Typed text is an overlay keyed by `id`, then `name`, then the selector used to type it. */
|
|
53
65
|
const keyOf = (selector: string, element: ElementSnapshot | undefined): string => {
|
|
54
66
|
if (element === undefined) return selector;
|
|
@@ -159,10 +171,11 @@ export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
|
|
|
159
171
|
return Promise.resolve(page.html);
|
|
160
172
|
},
|
|
161
173
|
query,
|
|
162
|
-
async click(selector: string
|
|
163
|
-
const element = await at(selector,
|
|
174
|
+
async click(selector: string): Promise<void> {
|
|
175
|
+
const element = await at(selector, 0);
|
|
164
176
|
if (element === undefined) throw fixtureMissing(`${page.url} ${selector}`, init.source);
|
|
165
|
-
const download =
|
|
177
|
+
const download =
|
|
178
|
+
recorded(page.downloads, selector) ?? recorded(page.downloads, element.attrs['id'] ?? '');
|
|
166
179
|
if (download !== undefined) armed = download;
|
|
167
180
|
const href =
|
|
168
181
|
element.attrs['data-goto'] ?? (element.tag === 'a' ? element.attrs['href'] : undefined);
|
|
@@ -181,12 +194,12 @@ export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
|
|
|
181
194
|
},
|
|
182
195
|
evaluate(expression: string): Promise<unknown> {
|
|
183
196
|
live();
|
|
184
|
-
const
|
|
197
|
+
const answer = recorded(page.evaluate, expression);
|
|
185
198
|
// Unrecorded and therefore refused, for the same reason an unrecorded page is: an offline
|
|
186
199
|
// driver that invented an answer here would make the assertion above it meaningless.
|
|
187
|
-
if (
|
|
200
|
+
if (answer === undefined)
|
|
188
201
|
throw fixtureMissing(`${page.url} evaluate(${expression})`, init.source);
|
|
189
|
-
return Promise.resolve(JSON.parse(
|
|
202
|
+
return Promise.resolve(JSON.parse(answer) as unknown);
|
|
190
203
|
},
|
|
191
204
|
screenshot: (_options: CaptureOptions): Promise<Uint8Array> => Promise.resolve(FAKE_PNG),
|
|
192
205
|
pdf: (_options: CaptureOptions): Promise<Uint8Array> => Promise.resolve(FAKE_PDF),
|
|
@@ -196,18 +209,20 @@ export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
|
|
|
196
209
|
session = next;
|
|
197
210
|
return Promise.resolve();
|
|
198
211
|
},
|
|
199
|
-
|
|
212
|
+
// `async`, so the refusal REJECTS: the method is typed `Promise<ScrapeDownloadFile>` and a
|
|
213
|
+
// synchronous `throw` from one escapes past `download().catch(…)` at every caller.
|
|
214
|
+
async download(options: { readonly timeoutMs: number }): Promise<ScrapeDownloadFile> {
|
|
200
215
|
if (armed === undefined) throw downloadTimeout(options.timeoutMs, page.url);
|
|
201
216
|
const { filename, contents } = splitDownload(armed);
|
|
202
217
|
armed = undefined;
|
|
203
|
-
return
|
|
218
|
+
return { filename, bytes: new TextEncoder().encode(contents) };
|
|
204
219
|
},
|
|
205
220
|
async frames(): Promise<readonly FrameRef[]> {
|
|
206
221
|
const refs: FrameRef[] = [];
|
|
207
222
|
for (const element of await queryHtml(page.html, 'iframe')) {
|
|
208
223
|
const name = element.attrs['name'] ?? element.attrs['id'] ?? '';
|
|
209
224
|
const src = element.attrs['src'] ?? '';
|
|
210
|
-
const html = page.frames
|
|
225
|
+
const html = recorded(page.frames, name) ?? recorded(page.frames, src);
|
|
211
226
|
if (html === undefined)
|
|
212
227
|
throw fixtureMissing(`${page.url} iframe ${name || src}`, init.source);
|
|
213
228
|
const url = src === '' ? page.url : new URL(src, page.url).toString();
|
package/src/http.ts
CHANGED
|
@@ -10,25 +10,57 @@
|
|
|
10
10
|
// guarantee the page vocabulary makes — and a different exit IP mid-session is exactly what
|
|
11
11
|
// anti-bot systems look for.
|
|
12
12
|
|
|
13
|
+
import { readWithinLimit } from '@ultimat3/core';
|
|
13
14
|
import type { StandardSchemaV1 } from '@ultimat3/schema';
|
|
14
15
|
import { parse } from '@ultimat3/schema';
|
|
15
16
|
import type { ScrapeClock } from './clock';
|
|
16
17
|
import { cookieHeaderFor } from './cookie-scope';
|
|
17
|
-
import { hostBlocked, httpFailed, scrapeTimeout } from './error-throws';
|
|
18
|
+
import { bodyTooLarge, hostBlocked, httpFailed, scrapeTimeout } from './error-throws';
|
|
18
19
|
import type { InterceptRules } from './intercept';
|
|
19
20
|
import { interceptVerdict } from './intercept';
|
|
20
21
|
import type { NetworkRing } from './rings';
|
|
21
22
|
import type { RobotsGate } from './robots';
|
|
22
23
|
import type { SessionSnapshot } from './session-state';
|
|
23
24
|
|
|
25
|
+
/**
|
|
26
|
+
* Just the call. `typeof fetch` also carries `preconnect`, which no test double and no app wrapper
|
|
27
|
+
* can supply — so an option typed `typeof fetch` was unusable without a double cast, which is
|
|
28
|
+
* exactly what every caller of it had written. The same seam `@ultimat3/cache`, `@ultimat3/auth`
|
|
29
|
+
* and `@ultimat3/mail` already name.
|
|
30
|
+
*/
|
|
31
|
+
export type ScrapeFetch = (input: string, init: ScrapeFetchInit) => Promise<Response>;
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* `RequestInit` plus the one Bun extension this package sets. Named rather than cast: the DOM's
|
|
35
|
+
* `RequestInit` has no `proxy`, and an `as RequestInit` over the literal silenced the excess-key
|
|
36
|
+
* check for `proxy` AND for every neighbouring key it was standing next to.
|
|
37
|
+
*/
|
|
38
|
+
export interface ScrapeFetchInit extends RequestInit {
|
|
39
|
+
/** The session's exit. A different exit IP mid-session is a different client to an anti-bot. */
|
|
40
|
+
readonly proxy?: string | undefined;
|
|
41
|
+
}
|
|
42
|
+
|
|
24
43
|
export interface HttpRequestInit {
|
|
25
44
|
readonly method?: string | undefined;
|
|
26
45
|
readonly headers?: Readonly<Record<string, string>> | undefined;
|
|
27
46
|
readonly body?: string | undefined;
|
|
28
47
|
/** Milliseconds. Falls back to the session's own default. */
|
|
29
48
|
readonly timeout?: number | undefined;
|
|
49
|
+
/**
|
|
50
|
+
* Response-body ceiling in bytes. Falls back to `DEFAULT_HTTP_MAX_BYTES`. Never absent: a
|
|
51
|
+
* deadline bounds time and a scraped endpoint is somebody else's, so the only thing standing
|
|
52
|
+
* between a hostile stream and the worker's heap is a number.
|
|
53
|
+
*/
|
|
54
|
+
readonly maxBytes?: number | undefined;
|
|
30
55
|
}
|
|
31
56
|
|
|
57
|
+
/**
|
|
58
|
+
* Generous for the JSON endpoint behind a paginated page — the reason this transport exists — and
|
|
59
|
+
* far under what OOM-kills a worker. Raised per call with `{ maxBytes }`, never globally: a run
|
|
60
|
+
* that genuinely pulls a large export says so at the call site that pulls it.
|
|
61
|
+
*/
|
|
62
|
+
export const DEFAULT_HTTP_MAX_BYTES = 32 * 1024 * 1024;
|
|
63
|
+
|
|
32
64
|
export interface ScrapeResponse {
|
|
33
65
|
readonly url: string;
|
|
34
66
|
readonly status: number;
|
|
@@ -66,13 +98,26 @@ export interface HttpTransportInit {
|
|
|
66
98
|
readonly onActivity?: (() => void) | undefined;
|
|
67
99
|
/** The SAME proxy the browser dialled through. A different exit IP is a different client. */
|
|
68
100
|
readonly proxy?: string | undefined;
|
|
69
|
-
readonly fetch?:
|
|
101
|
+
readonly fetch?: ScrapeFetch | undefined;
|
|
70
102
|
}
|
|
71
103
|
|
|
104
|
+
/**
|
|
105
|
+
* Response headers as data, on a NULL prototype and written with `defineProperty`.
|
|
106
|
+
*
|
|
107
|
+
* The header set on a scraping leg is entirely the site's. `out[key] = value` on a plain object
|
|
108
|
+
* DROPS `__proto__` — a legal HTTP field-name token — because the setter it hits refuses a string
|
|
109
|
+
* and files no own key, and it leaves `headers['toString']` answering a function the site never
|
|
110
|
+
* sent. Both make `Readonly<Record<string, string>>` a lie the caller cannot see through.
|
|
111
|
+
*/
|
|
72
112
|
const headerRecord = (headers: Headers): Record<string, string> => {
|
|
73
|
-
const out
|
|
113
|
+
const out = Object.create(null) as Record<string, string>;
|
|
74
114
|
headers.forEach((value, key) => {
|
|
75
|
-
out
|
|
115
|
+
Object.defineProperty(out, key, {
|
|
116
|
+
value,
|
|
117
|
+
enumerable: true,
|
|
118
|
+
writable: true,
|
|
119
|
+
configurable: true,
|
|
120
|
+
});
|
|
76
121
|
});
|
|
77
122
|
return out;
|
|
78
123
|
};
|
|
@@ -105,7 +150,7 @@ export function responseOver(
|
|
|
105
150
|
* robots rule, and neither is re-implemented for the second leg.
|
|
106
151
|
*/
|
|
107
152
|
export function httpOverFetch(init: HttpTransportInit): ScrapeHttp {
|
|
108
|
-
const call = init.fetch ?? fetch;
|
|
153
|
+
const call: ScrapeFetch = init.fetch ?? fetch;
|
|
109
154
|
return {
|
|
110
155
|
async request(url: string, request: HttpRequestInit = {}): Promise<ScrapeResponse> {
|
|
111
156
|
init.onActivity?.();
|
|
@@ -135,7 +180,7 @@ export function httpOverFetch(init: HttpTransportInit): ScrapeHttp {
|
|
|
135
180
|
...(request.body === undefined ? {} : { body: request.body }),
|
|
136
181
|
signal: AbortSignal.any(signals),
|
|
137
182
|
...(init.proxy === undefined ? {} : { proxy: init.proxy }),
|
|
138
|
-
}
|
|
183
|
+
});
|
|
139
184
|
init.network.push({
|
|
140
185
|
method: request.method ?? 'GET',
|
|
141
186
|
url,
|
|
@@ -143,7 +188,13 @@ export function httpOverFetch(init: HttpTransportInit): ScrapeHttp {
|
|
|
143
188
|
resourceType: 'fetch',
|
|
144
189
|
at: init.clock.now().getTime(),
|
|
145
190
|
});
|
|
146
|
-
|
|
191
|
+
// Counted as it arrives rather than `.text()`, which materialises first and checks never:
|
|
192
|
+
// a 30s stream at 50MB/s is a 1.5GB allocation the worker does not get back, and it takes
|
|
193
|
+
// every other job on that worker with it. The same read `robots-fetch.ts` performs.
|
|
194
|
+
const maxBytes = request.maxBytes ?? DEFAULT_HTTP_MAX_BYTES;
|
|
195
|
+
const capped = await readWithinLimit(response.body, maxBytes);
|
|
196
|
+
if ('over' in capped) throw bodyTooLarge(url, capped.over, maxBytes);
|
|
197
|
+
const body = new TextDecoder().decode(capped.bytes);
|
|
147
198
|
return responseOver(url, response.status, headerRecord(response.headers), () =>
|
|
148
199
|
Promise.resolve(body),
|
|
149
200
|
);
|
package/src/index.ts
CHANGED
|
@@ -5,7 +5,12 @@
|
|
|
5
5
|
export type { ActionabilityState, ActionabilityWait } from './actionability';
|
|
6
6
|
export { actionabilityProblem, awaitActionable, DEFAULT_POLL_MS, isStable } from './actionability';
|
|
7
7
|
export type { ArtifactRef, ArtifactWriter, ArtifactWriterInit } from './artifacts';
|
|
8
|
-
export {
|
|
8
|
+
export {
|
|
9
|
+
contentTypeFor,
|
|
10
|
+
createArtifactWriter,
|
|
11
|
+
DEFAULT_ARTIFACT_PREFIX,
|
|
12
|
+
DEFAULT_CONTENT_TYPE,
|
|
13
|
+
} from './artifacts';
|
|
9
14
|
export type {
|
|
10
15
|
AuthContext,
|
|
11
16
|
PromptHandler,
|
|
@@ -42,6 +47,7 @@ export { FIXTURE_DRIVER, fixtureBrowser, recordingFilename } from './driver-fixt
|
|
|
42
47
|
export {
|
|
43
48
|
authFailed,
|
|
44
49
|
blocked,
|
|
50
|
+
bodyTooLarge,
|
|
45
51
|
browserUnreachable,
|
|
46
52
|
cdpAttachFailed,
|
|
47
53
|
downloadTimeout,
|
|
@@ -97,7 +103,7 @@ export { markupRequests } from './html-requests';
|
|
|
97
103
|
export type { HtmlTargetInit, RecordingLookup } from './html-target';
|
|
98
104
|
export { htmlTarget } from './html-target';
|
|
99
105
|
export type { HttpRequestInit, HttpTransportInit, ScrapeHttp, ScrapeResponse } from './http';
|
|
100
|
-
export { httpOverFetch, responseOver } from './http';
|
|
106
|
+
export { DEFAULT_HTTP_MAX_BYTES, httpOverFetch, responseOver } from './http';
|
|
101
107
|
export type { HttpRecordingLookup, RecordedHttpInit } from './http-recorded';
|
|
102
108
|
export { httpRecordingFilename, httpRecordingsOf, recordedHttp } from './http-recorded';
|
|
103
109
|
export type { InterceptRules, InterceptVerdict } from './intercept';
|
|
@@ -137,6 +143,12 @@ export type {
|
|
|
137
143
|
export { createRing, DEFAULT_RING_CAPACITY, RESOURCE_TYPES } from './rings';
|
|
138
144
|
export type { RobotsFetch, RobotsGate, RobotsGateInit, RobotsPolicy, RobotsRules } from './robots';
|
|
139
145
|
export { createRobotsGate, DEFAULT_ROBOTS_AGENT, parseRobots, robotsAllows } from './robots';
|
|
146
|
+
export type { RobotsFetchInit } from './robots-fetch';
|
|
147
|
+
export {
|
|
148
|
+
DEFAULT_ROBOTS_MAX_BYTES,
|
|
149
|
+
DEFAULT_ROBOTS_TIMEOUT_MS,
|
|
150
|
+
robotsFetcher,
|
|
151
|
+
} from './robots-fetch';
|
|
140
152
|
export type {
|
|
141
153
|
ScrapeArtifacts,
|
|
142
154
|
ScrapeDefinition,
|
package/src/page-over-target.ts
CHANGED
|
@@ -105,7 +105,7 @@ function frameOver(
|
|
|
105
105
|
waitFor: (selector, options) => wait(selector, options, 'actionable'),
|
|
106
106
|
async click(selector, options): Promise<void> {
|
|
107
107
|
await wait(selector, options, 'actionable');
|
|
108
|
-
await (await resolve()).click(selector
|
|
108
|
+
await (await resolve()).click(selector);
|
|
109
109
|
},
|
|
110
110
|
async type(selector, text, options): Promise<void> {
|
|
111
111
|
await wait(selector, options, 'actionable');
|
|
@@ -187,8 +187,7 @@ export function pageOverTarget(target: ScrapeTarget, ctx: PageContext): ScrapePa
|
|
|
187
187
|
// in object storage, forever. Refused rather than masked — a mask over pixels is a guess
|
|
188
188
|
// about layout, and `page.html()` already gives a redacted artifact that is exact.
|
|
189
189
|
if (state.tainted) throw secretExposed(kind, target.url());
|
|
190
|
-
const
|
|
191
|
-
const request = { fullPage: options?.fullPage, timeoutMs };
|
|
190
|
+
const request = { fullPage: options?.fullPage };
|
|
192
191
|
return kind === 'screenshot' ? target.screenshot(request) : target.pdf(request);
|
|
193
192
|
};
|
|
194
193
|
return {
|
|
@@ -204,8 +203,12 @@ export function pageOverTarget(target: ScrapeTarget, ctx: PageContext): ScrapePa
|
|
|
204
203
|
},
|
|
205
204
|
screenshot: (options) => capture('screenshot', options),
|
|
206
205
|
pdf: (options) => capture('pdf', options),
|
|
207
|
-
|
|
208
|
-
|
|
206
|
+
// `async`, and that is the whole point of the keyword here: `ScrapeTarget` is the seam a third
|
|
207
|
+
// party implements, and a driver that THROWS from its promise-typed `download()` would escape
|
|
208
|
+
// past this page's caller `.catch()` if the forward were a bare arrow.
|
|
209
|
+
async download(options?: DownloadRequest): Promise<ScrapeDownloadFile> {
|
|
210
|
+
return await target.download({ timeoutMs: options?.timeout ?? ctx.defaultTimeoutMs });
|
|
211
|
+
},
|
|
209
212
|
cookies: (): Promise<readonly ScrapeCookie[]> => target.cookies(),
|
|
210
213
|
session: () => target.session(),
|
|
211
214
|
console: () => target.console.entries(),
|
package/src/page.ts
CHANGED
|
@@ -64,9 +64,9 @@ export interface ScrapeFrame {
|
|
|
64
64
|
frame(nameOrSelector: string): ScrapeFrame;
|
|
65
65
|
}
|
|
66
66
|
|
|
67
|
+
/** `fullPage` only. `timeout` is gone with the port's — see `CaptureOptions` for why. */
|
|
67
68
|
export interface CaptureRequest {
|
|
68
69
|
readonly fullPage?: boolean | undefined;
|
|
69
|
-
readonly timeout?: number | undefined;
|
|
70
70
|
}
|
|
71
71
|
|
|
72
72
|
export interface DownloadRequest {
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
// The ONE `/robots.txt` read the gate performs when the caller injects no `fetchText`.
|
|
2
|
+
//
|
|
3
|
+
// It exists as its own file because the production default was the only network call in this
|
|
4
|
+
// package with no deadline, no size cap and no proxy — and `scrape-run.ts` builds the gate with no
|
|
5
|
+
// `fetchText`, so production always took it. Every existing gate test injected one, which is how a
|
|
6
|
+
// read that could park a run forever stayed green.
|
|
7
|
+
//
|
|
8
|
+
// The exit is a RESOLVER, not a string: `scrape-run.ts` builds this gate as an argument to
|
|
9
|
+
// `driver.open()`, and the proxy is a driver option the session only reports on the way back out.
|
|
10
|
+
|
|
11
|
+
import { readWithinLimit } from '@ultimat3/core';
|
|
12
|
+
import type { ScrapeFetch } from './http';
|
|
13
|
+
import type { RobotsFetch } from './robots';
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* A deadline is applied ALWAYS, proxy or no proxy, session or no session: the failure it prevents
|
|
17
|
+
* is a hung origin whose cached promise then parks every later navigation to that origin, past
|
|
18
|
+
* `ctx.signal`, the watchdog and the job timeout. Ten seconds is long for a static text file and
|
|
19
|
+
* short against a slow-loris.
|
|
20
|
+
*/
|
|
21
|
+
export const DEFAULT_ROBOTS_TIMEOUT_MS = 10_000;
|
|
22
|
+
|
|
23
|
+
/** Google's own documented ceiling for the file, and generous for a list of path prefixes. */
|
|
24
|
+
export const DEFAULT_ROBOTS_MAX_BYTES = 500 * 1024;
|
|
25
|
+
|
|
26
|
+
export interface RobotsFetchInit {
|
|
27
|
+
/** Per-read wall clock. Defaults to `DEFAULT_ROBOTS_TIMEOUT_MS`. */
|
|
28
|
+
readonly timeoutMs?: number | undefined;
|
|
29
|
+
/** The run's cancellation, when there is one. Composed with the deadline, never replacing it. */
|
|
30
|
+
readonly signal?: AbortSignal | undefined;
|
|
31
|
+
/**
|
|
32
|
+
* The SAME proxy the browser dialled through, when the session has one — asked PER READ, never
|
|
33
|
+
* captured. A resolver rather than a string because construction order forbids the string: the
|
|
34
|
+
* gate is an argument to `driver.open()` and the proxy is a driver option resolved inside it,
|
|
35
|
+
* so a value passed here could only ever be the one nobody has yet. That is how the robots read
|
|
36
|
+
* came to exit from the worker's IP while every page load exited through the proxy — and how an
|
|
37
|
+
* origin reachable ONLY through the proxy read as "no robots.txt", which is allow-everything.
|
|
38
|
+
*
|
|
39
|
+
* Optional by design: proxies are an opt-in leg, and an origin reachable directly must still be
|
|
40
|
+
* asked for its rules.
|
|
41
|
+
*/
|
|
42
|
+
readonly proxy?: (() => string | undefined) | undefined;
|
|
43
|
+
readonly maxBytes?: number | undefined;
|
|
44
|
+
/**
|
|
45
|
+
* The platform `fetch`, injectable so the default path itself is testable. `ScrapeFetch` and not
|
|
46
|
+
* `typeof fetch`: the latter also carries `preconnect`, so nothing a caller can write satisfies
|
|
47
|
+
* it and the option was reachable only through a cast.
|
|
48
|
+
*/
|
|
49
|
+
readonly fetch?: ScrapeFetch | undefined;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Reads `robotsUrl`, or answers `undefined` — which the gate reads as "no restrictions", the
|
|
54
|
+
* standard's own answer for a file it cannot obtain. A deadline that fires, a body past the cap
|
|
55
|
+
* and a 404 are all the same answer on purpose: none of them is evidence of a rule.
|
|
56
|
+
*/
|
|
57
|
+
export function robotsFetcher(init: RobotsFetchInit = {}): RobotsFetch {
|
|
58
|
+
const call: ScrapeFetch = init.fetch ?? fetch;
|
|
59
|
+
const limit = init.maxBytes ?? DEFAULT_ROBOTS_MAX_BYTES;
|
|
60
|
+
return async (robotsUrl: string): Promise<string | undefined> => {
|
|
61
|
+
// Armed per read, not per gate: the gate is long-lived and reads once per origin, so a
|
|
62
|
+
// deadline created alongside it would already have expired by the second origin.
|
|
63
|
+
const deadline = AbortSignal.timeout(init.timeoutMs ?? DEFAULT_ROBOTS_TIMEOUT_MS);
|
|
64
|
+
const signal = init.signal === undefined ? deadline : AbortSignal.any([deadline, init.signal]);
|
|
65
|
+
// Resolved here, at the read, because the session that owns the exit did not exist when this
|
|
66
|
+
// fetcher was built. An empty string is not an exit and is dropped with the absent one.
|
|
67
|
+
const proxy = init.proxy?.();
|
|
68
|
+
try {
|
|
69
|
+
const response = await call(robotsUrl, {
|
|
70
|
+
signal,
|
|
71
|
+
...(proxy === undefined || proxy === '' ? {} : { proxy }),
|
|
72
|
+
});
|
|
73
|
+
if (!response.ok) return undefined;
|
|
74
|
+
// Counted as it arrives rather than `.text()`, which materialises the whole body first: a
|
|
75
|
+
// multi-gigabyte robots.txt is a heap the worker never gets back.
|
|
76
|
+
const capped = await readWithinLimit(response.body, limit);
|
|
77
|
+
return 'over' in capped ? undefined : new TextDecoder().decode(capped.bytes);
|
|
78
|
+
} catch {
|
|
79
|
+
return undefined;
|
|
80
|
+
}
|
|
81
|
+
};
|
|
82
|
+
}
|
package/src/robots.ts
CHANGED
|
@@ -6,6 +6,8 @@
|
|
|
6
6
|
// There is no boolean, because `robots: false` is a decision with no author.
|
|
7
7
|
|
|
8
8
|
import { robotsDisallowed } from './error-throws';
|
|
9
|
+
import type { RobotsFetchInit } from './robots-fetch';
|
|
10
|
+
import { robotsFetcher } from './robots-fetch';
|
|
9
11
|
|
|
10
12
|
export type RobotsPolicy = 'obey' | { readonly ignore: string };
|
|
11
13
|
|
|
@@ -56,14 +58,47 @@ export function parseRobots(text: string, agent: string): RobotsRules {
|
|
|
56
58
|
return { rules: groups.get(wanted) ?? groups.get('*') ?? [] };
|
|
57
59
|
}
|
|
58
60
|
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
61
|
+
/**
|
|
62
|
+
* `/private/*.pdf$` — the two wildcards robots.txt defines, and no others, matched by WALKING the
|
|
63
|
+
* pattern rather than by compiling one.
|
|
64
|
+
*
|
|
65
|
+
* The rule text is remote: robots.txt belongs to the site being scraped, and `robotsAllows` runs on
|
|
66
|
+
* every navigation and every HTTP-leg request, synchronously, on the worker's only thread. A regex
|
|
67
|
+
* built as `body.split('*').join('.*')` backtracks catastrophically on a non-matching path — a rule
|
|
68
|
+
* with 24 wildcards did not return inside 60s, past `ctx.signal`, the watchdog and the job timeout,
|
|
69
|
+
* all of which are downstream of a `return` that never happens. This walk is O(pattern × path) with
|
|
70
|
+
* no backtracking beyond the LAST star, so the class is removed rather than bounded.
|
|
71
|
+
*/
|
|
62
72
|
const patternMatches = (pattern: string, path: string): boolean => {
|
|
63
73
|
const anchored = pattern.endsWith('$');
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
74
|
+
// Unanchored means "matches a PREFIX of the path", which is the same statement as a full match
|
|
75
|
+
// against the pattern with one more `*` on the end — one code path instead of two.
|
|
76
|
+
const body = anchored ? pattern.slice(0, -1) : `${pattern}*`;
|
|
77
|
+
let p = 0;
|
|
78
|
+
let s = 0;
|
|
79
|
+
let star = -1;
|
|
80
|
+
let resume = 0;
|
|
81
|
+
while (s < path.length) {
|
|
82
|
+
if (p < body.length && body[p] === '*') {
|
|
83
|
+
star = p;
|
|
84
|
+
p += 1;
|
|
85
|
+
resume = s;
|
|
86
|
+
continue;
|
|
87
|
+
}
|
|
88
|
+
if (p < body.length && body[p] === path[s]) {
|
|
89
|
+
p += 1;
|
|
90
|
+
s += 1;
|
|
91
|
+
continue;
|
|
92
|
+
}
|
|
93
|
+
if (star === -1) return false;
|
|
94
|
+
// Only the most recent star is ever retried, which is what keeps this linear per star instead
|
|
95
|
+
// of exponential across all of them.
|
|
96
|
+
p = star + 1;
|
|
97
|
+
resume += 1;
|
|
98
|
+
s = resume;
|
|
99
|
+
}
|
|
100
|
+
while (p < body.length && body[p] === '*') p += 1;
|
|
101
|
+
return p === body.length;
|
|
67
102
|
};
|
|
68
103
|
|
|
69
104
|
/**
|
|
@@ -93,14 +128,10 @@ export interface RobotsGate {
|
|
|
93
128
|
|
|
94
129
|
export type RobotsFetch = (robotsUrl: string) => Promise<string | undefined>;
|
|
95
130
|
|
|
96
|
-
|
|
97
|
-
const response = await fetch(robotsUrl);
|
|
98
|
-
return response.ok ? await response.text() : undefined;
|
|
99
|
-
};
|
|
100
|
-
|
|
101
|
-
export interface RobotsGateInit {
|
|
131
|
+
export interface RobotsGateInit extends RobotsFetchInit {
|
|
102
132
|
readonly policy: RobotsPolicy;
|
|
103
133
|
readonly agent?: string | undefined;
|
|
134
|
+
/** A caller-supplied read. With none, `robotsFetcher` builds the deadlined, capped default. */
|
|
104
135
|
readonly fetchText?: RobotsFetch | undefined;
|
|
105
136
|
}
|
|
106
137
|
|
|
@@ -116,7 +147,7 @@ export function createRobotsGate(init: RobotsGateInit): RobotsGate {
|
|
|
116
147
|
return { assertAllowed: () => Promise.resolve(), ignoredBecause: init.policy.ignore };
|
|
117
148
|
}
|
|
118
149
|
const agent = init.agent ?? DEFAULT_ROBOTS_AGENT;
|
|
119
|
-
const read = init.fetchText ??
|
|
150
|
+
const read = init.fetchText ?? robotsFetcher(init);
|
|
120
151
|
const cache = new Map<string, Promise<RobotsRules>>();
|
|
121
152
|
return {
|
|
122
153
|
ignoredBecause: undefined,
|
package/src/scrape-run.ts
CHANGED
|
@@ -94,18 +94,36 @@ export async function runScrape<I, Row>(
|
|
|
94
94
|
// Read BEFORE the browser opens: a refused credential must not reach a login form again, and
|
|
95
95
|
// opening a session first would already have spent an identity on a run that cannot succeed.
|
|
96
96
|
const restored = await restorableSession(plan);
|
|
97
|
+
const pageTimeoutMs = toMillis(definition.pageTimeout, DEFAULT_PAGE_TIMEOUT_MS);
|
|
98
|
+
// The exit the session dials, readable only AFTER `driver.open()` — the proxy is a driver
|
|
99
|
+
// option and the gate below is an argument to `open()`, so the gate asks for it per read
|
|
100
|
+
// instead of being handed a value that cannot exist yet. Every read happens during a
|
|
101
|
+
// navigation, which is after this is assigned.
|
|
102
|
+
let sessionProxy: string | undefined;
|
|
97
103
|
const session = await driver.open({
|
|
98
104
|
name: definition.name,
|
|
99
105
|
rules,
|
|
100
106
|
clock,
|
|
101
|
-
timeoutMs:
|
|
107
|
+
timeoutMs: pageTimeoutMs,
|
|
102
108
|
secrets,
|
|
103
|
-
|
|
109
|
+
// The gate reads `/robots.txt` over the network, so it gets the run's deadline, the run's
|
|
110
|
+
// cancellation and the run's exit, like every other call this package makes. Without the
|
|
111
|
+
// first two a hung origin parks every later navigation to it on one cached promise,
|
|
112
|
+
// unreachable by `ctx.signal`; without the third the read leaves from a different IP than
|
|
113
|
+
// every page load, and an origin reachable only through the proxy answers nothing — which
|
|
114
|
+
// this gate reads as "no restrictions".
|
|
115
|
+
robots: createRobotsGate({
|
|
116
|
+
policy: definition.robots ?? 'obey',
|
|
117
|
+
timeoutMs: pageTimeoutMs,
|
|
118
|
+
signal: args.ctx.signal,
|
|
119
|
+
proxy: () => sessionProxy,
|
|
120
|
+
}),
|
|
104
121
|
signal: args.ctx.signal,
|
|
105
122
|
restore: restored,
|
|
106
123
|
pace: (signal) => pace(signal),
|
|
107
124
|
watchdog: definition.watchdog,
|
|
108
125
|
});
|
|
126
|
+
sessionProxy = session.proxy;
|
|
109
127
|
|
|
110
128
|
try {
|
|
111
129
|
if (definition.auth !== undefined) {
|
|
@@ -155,8 +173,11 @@ export async function runScrape<I, Row>(
|
|
|
155
173
|
};
|
|
156
174
|
} catch (thrown) {
|
|
157
175
|
logger.error('scrape.failed', { code: errorCode(thrown) });
|
|
158
|
-
if (errorCode(thrown) === 'X_SCRAPE_AUTH_FAILED')
|
|
159
|
-
|
|
176
|
+
if (errorCode(thrown) === 'X_SCRAPE_AUTH_FAILED') {
|
|
177
|
+
await recordSessionOutcome('session.refuse', logger, () => markRefused(plan));
|
|
178
|
+
} else if (burnsSession(thrown)) {
|
|
179
|
+
await recordSessionOutcome('session.burn', logger, () => burnSession(plan));
|
|
180
|
+
}
|
|
160
181
|
if (definition.artifacts?.onFailure !== false) await saveFailureArtifact(session, artifact);
|
|
161
182
|
throw thrown;
|
|
162
183
|
} finally {
|
|
@@ -165,6 +186,29 @@ export async function runScrape<I, Row>(
|
|
|
165
186
|
}
|
|
166
187
|
}
|
|
167
188
|
|
|
189
|
+
/**
|
|
190
|
+
* The tombstone or the burn, on the way out — best effort, and it may NEVER replace the failure
|
|
191
|
+
* that caused it. `markRefused` reaches `store.save()` reaches `storage.put()`, so an S3 503 or an
|
|
192
|
+
* `X_STORAGE_PATH_UNSAFE` from a tenant whose key sanitises to nothing used to propagate out of
|
|
193
|
+
* the catch and REPLACE a terminal `X_SCRAPE_AUTH_FAILED` with a retryable one. Attempt 2 then
|
|
194
|
+
* found no tombstone — the save is what failed — and walked the same rejected password back to
|
|
195
|
+
* the login form; attempt 3 locks the account. Same rule as `saveFailureArtifact`, and the same
|
|
196
|
+
* reason: the run's own error is the one the reader needs.
|
|
197
|
+
*/
|
|
198
|
+
async function recordSessionOutcome(
|
|
199
|
+
step: 'session.refuse' | 'session.burn',
|
|
200
|
+
logger: ReturnType<typeof scrapeLogger>,
|
|
201
|
+
write: () => Promise<void>,
|
|
202
|
+
): Promise<void> {
|
|
203
|
+
try {
|
|
204
|
+
await write();
|
|
205
|
+
} catch (thrown) {
|
|
206
|
+
// Logged rather than swallowed silently: the tombstone is missing, so the NEXT attempt will
|
|
207
|
+
// re-probe rather than refuse cheaply, and that is a fact an operator has to be able to see.
|
|
208
|
+
logger.error('scrape.session.write_failed', { step, code: errorCode(thrown) });
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
|
|
168
212
|
/**
|
|
169
213
|
* The body, with at most ONE recovery pass. `recover` never runs for a failure that must not be
|
|
170
214
|
* retried — a rejected credential is the case, and asking a model to "fix" a wrong password is
|
package/src/scrape.ts
CHANGED
|
@@ -18,6 +18,7 @@ import type { ArtifactWriter } from './artifacts';
|
|
|
18
18
|
import type { PromptHandler, ScrapeAuth } from './auth';
|
|
19
19
|
import type { ScrapeClock } from './clock';
|
|
20
20
|
import type { ScrapeDriver } from './driver';
|
|
21
|
+
import { yieldHistoryMissing } from './error-throws';
|
|
21
22
|
import type { YieldExpectation, YieldHistory } from './expect';
|
|
22
23
|
import type { HostRule } from './hosts';
|
|
23
24
|
import type { ScrapeHttp } from './http';
|
|
@@ -137,6 +138,12 @@ export function scrape<I, Row>(definition: ScrapeDefinition<I, Row>): JobHandle<
|
|
|
137
138
|
`scrape "${definition.name}" declares rate: ${String(definition.rate)} — a rate is navigations per second, greater than zero`,
|
|
138
139
|
`set rate: 1 on scrape("${definition.name}"), or leave it out — to go faster raise the number, there is no unpaced mode`,
|
|
139
140
|
);
|
|
141
|
+
// Two halves that must be set together. `maxDrop` is a fraction of a trailing median and only
|
|
142
|
+
// `history:` can supply one, so declaring it alone is an alarm that cannot fire — refused here,
|
|
143
|
+
// where it is written, rather than discovered as a scrape that never once went red.
|
|
144
|
+
if (definition.expect?.maxDrop !== undefined && definition.history === undefined) {
|
|
145
|
+
throw yieldHistoryMissing(definition.name);
|
|
146
|
+
}
|
|
140
147
|
return job<I>({
|
|
141
148
|
name: definition.name,
|
|
142
149
|
input: definition.input,
|
package/src/secrets.ts
CHANGED
|
@@ -78,11 +78,25 @@ export function redactSecrets(text: string, secrets: ScrapeSecrets | undefined):
|
|
|
78
78
|
return out;
|
|
79
79
|
}
|
|
80
80
|
|
|
81
|
-
/** `<input
|
|
81
|
+
/** Every `<input …>` tag, whole, so the rewrite below never has to reason about attribute order. */
|
|
82
|
+
const INPUT_TAG = /<input\b[^>]*>/gi;
|
|
83
|
+
/** `type=password`, quoted either way or bare. */
|
|
84
|
+
const PASSWORD_TYPE = /\btype\s*=\s*(?:"password"|'password'|password)(?=[\s/>])/i;
|
|
85
|
+
const VALUE_ATTR = /\bvalue\s*=\s*(?:"[^"]*"|'[^']*'|[^\s>]*)/gi;
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* `<input type="password" value="hunter2">` -> `value=""`, whatever the value happened to be —
|
|
89
|
+
* and whatever ORDER the site wrote the attributes in.
|
|
90
|
+
*
|
|
91
|
+
* One regex over the whole tag required `type` to precede `value`, so `<input value="hunter2"
|
|
92
|
+
* type="password">` came through untouched. Attribute order is the site's choice, and
|
|
93
|
+
* `saveFailureArtifact` writes `page.html()` to object storage on every failed run: a
|
|
94
|
+
* server-rendered password on a reversed-attribute form was durably persisted. So: match the TAG
|
|
95
|
+
* first, then rewrite `value` inside it.
|
|
96
|
+
*/
|
|
82
97
|
export function blankPasswordFields(html: string): string {
|
|
83
|
-
return html.replaceAll(
|
|
84
|
-
|
|
85
|
-
'$1value=""',
|
|
98
|
+
return html.replaceAll(INPUT_TAG, (tag) =>
|
|
99
|
+
PASSWORD_TYPE.test(tag) ? tag.replaceAll(VALUE_ATTR, 'value=""') : tag,
|
|
86
100
|
);
|
|
87
101
|
}
|
|
88
102
|
|
package/src/session-state.ts
CHANGED
|
@@ -157,11 +157,35 @@ export function storageSessionStore(
|
|
|
157
157
|
};
|
|
158
158
|
}
|
|
159
159
|
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
160
|
+
/**
|
|
161
|
+
* A stored cookie is somebody else's JSON. `name` and `value` are what makes it a cookie at all;
|
|
162
|
+
* the four scope fields `ScrapeCookie` REQUIRES are completed here rather than asserted.
|
|
163
|
+
*
|
|
164
|
+
* Asserting them was the bug: this was a `value is ScrapeCookie` predicate that checked two of
|
|
165
|
+
* that type's six required fields, so a stored `{ name, value }` left `parseSessionState` typed as
|
|
166
|
+
* a whole cookie with no `domain` — and `cookieHeaderFor`, a public export, hands it to
|
|
167
|
+
* `cookieDomainMatches`, which calls `.trim()` on it and throws a bare `TypeError`.
|
|
168
|
+
*
|
|
169
|
+
* The defaults are the ones `cookie-scope.ts` already documents. An empty `domain` matches NO
|
|
170
|
+
* host, which is the point: an unscoped cookie must reach nothing, because the only other way to
|
|
171
|
+
* scope it is to infer the domain from whichever URL is asking, and that is exactly how a
|
|
172
|
+
* `bank.test` session cookie reaches `evilbank.test`. `/` is §5.1.4's reading of an absent path,
|
|
173
|
+
* and an attribute a jar never wrote is `false`.
|
|
174
|
+
*/
|
|
175
|
+
const toCookie = (value: unknown): ScrapeCookie | undefined => {
|
|
176
|
+
if (typeof value !== 'object' || value === null) return undefined;
|
|
177
|
+
const entry = value as Partial<ScrapeCookie>;
|
|
178
|
+
if (typeof entry.name !== 'string' || typeof entry.value !== 'string') return undefined;
|
|
179
|
+
return {
|
|
180
|
+
name: entry.name,
|
|
181
|
+
value: entry.value,
|
|
182
|
+
domain: typeof entry.domain === 'string' ? entry.domain : '',
|
|
183
|
+
path: typeof entry.path === 'string' ? entry.path : '/',
|
|
184
|
+
...(typeof entry.expires === 'number' ? { expires: entry.expires } : {}),
|
|
185
|
+
httpOnly: entry.httpOnly === true,
|
|
186
|
+
secure: entry.secure === true,
|
|
187
|
+
};
|
|
188
|
+
};
|
|
165
189
|
|
|
166
190
|
/** Stored JSON is `unknown`. Read structurally, and answer `undefined` rather than half a session. */
|
|
167
191
|
export function parseSessionState(raw: unknown, key: string): SessionState | undefined {
|
|
@@ -172,7 +196,7 @@ export function parseSessionState(raw: unknown, key: string): SessionState | und
|
|
|
172
196
|
key,
|
|
173
197
|
savedAt: value.savedAt,
|
|
174
198
|
...(typeof value.refusedAt === 'string' ? { refusedAt: value.refusedAt } : {}),
|
|
175
|
-
cookies: value.cookies.
|
|
199
|
+
cookies: value.cookies.flatMap((cookie: unknown) => toCookie(cookie) ?? []),
|
|
176
200
|
headers: value.headers ?? {},
|
|
177
201
|
storage: value.storage ?? {},
|
|
178
202
|
userAgent: value.userAgent ?? '',
|
package/src/target.ts
CHANGED
|
@@ -74,9 +74,17 @@ export interface GotoOptions {
|
|
|
74
74
|
readonly signal?: AbortSignal | undefined;
|
|
75
75
|
}
|
|
76
76
|
|
|
77
|
+
/**
|
|
78
|
+
* `fullPage` and nothing else. It carried a required `timeoutMs` until 2026-08 that NO driver
|
|
79
|
+
* honoured — `cdp-target.ts` read only `fullPage`, `html-target.ts` ignored the whole object —
|
|
80
|
+
* so `page.screenshot({ timeout })` was a documented deadline that bounded nothing. Deleted
|
|
81
|
+
* rather than implemented: the CDP port's own `screenshot({ fullPage })` has no timeout slot to
|
|
82
|
+
* forward it to, and a deadline enforced in `page-over-target.ts` would have to race
|
|
83
|
+
* `ScrapeClock.sleep`, which under `testClock` resolves on the first microtask and would time
|
|
84
|
+
* out every capture in every test. A driver's own default is the honest bound.
|
|
85
|
+
*/
|
|
77
86
|
export interface CaptureOptions {
|
|
78
87
|
readonly fullPage?: boolean | undefined;
|
|
79
|
-
readonly timeoutMs: number;
|
|
80
88
|
}
|
|
81
89
|
|
|
82
90
|
/**
|
|
@@ -93,7 +101,16 @@ export interface ScrapeTarget {
|
|
|
93
101
|
/** Serialised HTML of THIS target — the document for a page, the subtree for a frame. */
|
|
94
102
|
content(): Promise<string>;
|
|
95
103
|
query(selector: string): Promise<readonly ElementSnapshot[]>;
|
|
96
|
-
|
|
104
|
+
/**
|
|
105
|
+
* Clicks the FIRST match. It took an `index` until 2026-08 that `html-target.ts` honoured and
|
|
106
|
+
* `cdp-target.ts` dropped — its implementations are `click: (selector) => …`, so puppeteer
|
|
107
|
+
* clicked match 0 whatever was asked. They agreed only because `page-over-target.ts`, the sole
|
|
108
|
+
* caller, always passed `0`, and no public vocabulary could set it: `ScrapeFrame.click` takes
|
|
109
|
+
* `(selector, options?: WaitOptions)`. A port member no app can reach and one driver ignores is
|
|
110
|
+
* a divergence waiting to be found by an app, so it is gone. `driver-parity.test.ts` pins that
|
|
111
|
+
* all three drivers click the first match.
|
|
112
|
+
*/
|
|
113
|
+
click(selector: string): Promise<void>;
|
|
97
114
|
/** Appends, exactly as typing does. Clearing first is `fill`'s job at the page level. */
|
|
98
115
|
type(selector: string, text: string): Promise<void>;
|
|
99
116
|
clear(selector: string): Promise<void>;
|