@ultimat3/scraping 2.0.0 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ultimat3/scraping",
3
- "version": "2.0.0",
3
+ "version": "4.0.0",
4
4
  "description": "Browser automation as a job: scrape() returns a JobHandle",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -30,9 +30,9 @@
30
30
  "test": "bun test"
31
31
  },
32
32
  "dependencies": {
33
- "@ultimat3/core": "2.0.0",
34
- "@ultimat3/jobs": "2.0.0",
35
- "@ultimat3/schema": "2.0.0",
36
- "@ultimat3/storage": "2.0.0"
33
+ "@ultimat3/core": "4.0.0",
34
+ "@ultimat3/jobs": "4.0.0",
35
+ "@ultimat3/schema": "4.0.0",
36
+ "@ultimat3/storage": "4.0.0"
37
37
  }
38
38
  }
package/src/artifacts.ts CHANGED
@@ -30,17 +30,26 @@ export interface ArtifactWriterInit {
30
30
 
31
31
  export const DEFAULT_ARTIFACT_PREFIX = 'scrape';
32
32
 
33
- const CONTENT_TYPES: Readonly<Record<string, string>> = {
34
- html: 'text/html; charset=utf-8',
35
- png: 'image/png',
36
- pdf: 'application/pdf',
37
- json: 'application/json',
38
- csv: 'text/csv',
39
- txt: 'text/plain; charset=utf-8',
40
- };
33
+ export const DEFAULT_CONTENT_TYPE = 'application/octet-stream';
34
+
35
+ /**
36
+ * A `Map`, not an object literal — the same choice `failures.ts` makes, for the same reason. The
37
+ * extension comes off a caller-supplied filename, and on a download that filename came off the
38
+ * site's own `Content-Disposition`: indexing a plain object with it answered a FUNCTION for
39
+ * `report.constructor` and an object for `report.__proto__`, where the type says `string`, and
40
+ * that non-string went on to be `ref.contentType` and an S3 header.
41
+ */
42
+ const CONTENT_TYPES: ReadonlyMap<string, string> = new Map([
43
+ ['html', 'text/html; charset=utf-8'],
44
+ ['png', 'image/png'],
45
+ ['pdf', 'application/pdf'],
46
+ ['json', 'application/json'],
47
+ ['csv', 'text/csv'],
48
+ ['txt', 'text/plain; charset=utf-8'],
49
+ ]);
41
50
 
42
51
  export const contentTypeFor = (name: string): string =>
43
- CONTENT_TYPES[name.split('.').pop()?.toLowerCase() ?? ''] ?? 'application/octet-stream';
52
+ CONTENT_TYPES.get(name.split('.').pop()?.toLowerCase() ?? '') ?? DEFAULT_CONTENT_TYPE;
44
53
 
45
54
  /**
46
55
  * A writer with no storage driver is a NO-OP that still answers a key — never a throw. An app
package/src/cdp-target.ts CHANGED
@@ -1,6 +1,7 @@
1
1
  // `ScrapeTarget` over a real browser, through the structural CDP port. Everything driver-specific
2
2
  // in this package lives here and in `driver-cdp.ts`; the vocabulary above it does not change.
3
3
 
4
+ import { isUltimateError } from '@ultimat3/core';
4
5
  import type { StandardSchemaV1 } from '@ultimat3/schema';
5
6
  import { parse, t } from '@ultimat3/schema';
6
7
  import type { CdpBrowserLike, CdpFrameLike, CdpPageLike, CdpRequestLike } from './cdp-port';
@@ -87,6 +88,26 @@ const readStringFrom = (owner: unknown, key: string): string | undefined => {
87
88
  return typeof answer === 'string' ? answer : undefined;
88
89
  };
89
90
 
91
+ /**
92
+ * The one failure `guard()` must NOT re-label, and the line is drawn at exactly one code.
93
+ *
94
+ * `X_NOT_IMPLEMENTED` is the only code that says "this build does not have the feature" — a fact
95
+ * about the launcher's own shape, never about the connection. A browser cannot produce it; only
96
+ * this file's own `scrapeNotImplemented()` can, from inside a guarded closure. Re-labelled as
97
+ * `X_SCRAPE_BROWSER_UNREACHABLE` (registered `retryable` in `errors.ts`) it spends every attempt
98
+ * in the scrape's retry policy on a method that is still missing on attempt five, and tells the
99
+ * operator the browser went away while the browser is answering fine.
100
+ *
101
+ * Every OTHER coded error stays wrapped, deliberately. `thrown instanceof UltimateError` is the
102
+ * naive version of this check and it is wrong: an `X_SCRAPE_TIMEOUT` raised while the socket was
103
+ * already dead would then arrive unwrapped, and "the browser went away mid-run" is the frame that
104
+ * makes a disconnect legible — which is the whole reason this wrapper exists.
105
+ *
106
+ * `isUltimateError`, not `instanceof`: the brand survives a duplicated module instance.
107
+ */
108
+ const isStructuralRefusal = (thrown: unknown): boolean =>
109
+ isUltimateError(thrown) && thrown.code === 'X_NOT_IMPLEMENTED';
110
+
90
111
  export interface CdpTargetInit {
91
112
  readonly page: CdpPageLike;
92
113
  readonly browser: CdpBrowserLike;
@@ -185,6 +206,7 @@ export async function cdpTarget(init: CdpTargetInit): Promise<ScrapeTarget> {
185
206
  return await run();
186
207
  } catch (thrown) {
187
208
  live();
209
+ if (isStructuralRefusal(thrown)) throw thrown;
188
210
  throw browserUnreachable(`${CDP_DRIVER} ${what}`, thrown);
189
211
  }
190
212
  };
@@ -256,15 +278,20 @@ export async function cdpTarget(init: CdpTargetInit): Promise<ScrapeTarget> {
256
278
  }
257
279
  return parse(cookieSchema, await source.cookies());
258
280
  }),
259
- download: (_options): Promise<ScrapeDownloadFile> => {
260
- // Honest stub, in the shape `packages/jobs/src/driver-redis.ts` uses. A real one needs
261
- // `Browser.setDownloadBehavior` over a raw CDP session plus a directory watch, and a
262
- // half-written version that silently returned empty bytes is worse than this line.
263
- throw scrapeNotImplemented(
264
- 'download() on the puppeteer driver',
265
- 'fetch the file inside the page — page.evaluate("fetch(url).then(r => r.text())") — or run this scrape on fixtureBrowser(), whose download() is complete',
266
- );
267
- },
281
+ // Honest stub, in the shape `packages/jobs/src/driver-redis.ts` uses. A real one needs
282
+ // `Browser.setDownloadBehavior` over a raw CDP session plus a directory watch, and a
283
+ // half-written version that silently returned empty bytes is worse than this line.
284
+ //
285
+ // REJECTS rather than throws: the method is typed `Promise<ScrapeDownloadFile>` and every
286
+ // caller of a promise-typed method handles its failure with `.catch()` or an `await` inside a
287
+ // `try` — a synchronous `throw` jumps over the first of those entirely.
288
+ download: (_options): Promise<ScrapeDownloadFile> =>
289
+ Promise.reject(
290
+ scrapeNotImplemented(
291
+ 'download() on the puppeteer driver',
292
+ 'fetch the file inside the page — page.evaluate("fetch(url).then(r => r.text())") — or run this scrape on fixtureBrowser(), whose download() is complete',
293
+ ),
294
+ ),
268
295
  frames: () =>
269
296
  guard('frames', () =>
270
297
  Promise.resolve(
package/src/driver-cdp.ts CHANGED
@@ -67,6 +67,19 @@ async function opened(browser: CdpBrowserLike, init: SessionInit): Promise<Scrap
67
67
  }
68
68
  }
69
69
 
70
+ /**
71
+ * The run's cancellation and the watchdog's, as ONE signal handed to every wait.
72
+ *
73
+ * The guard's abort half had no reader: both legs below were passed `init.signal`, so the
74
+ * watchdog's only production effect was `kill()` — and `Browser.process()` answers `null` for a
75
+ * browser obtained through `connect()`, which is `remoteBrowser()`, this file's primary path. A
76
+ * wedge therefore killed nothing and aborted a signal nobody composed, and the blocked await on
77
+ * the CDP socket stayed blocked past `ctx.signal` and past the watchdog. Verbatim incident #1 in
78
+ * `watchdog.ts`, unfixed for the attach path until the composition below.
79
+ */
80
+ const withWedgeSignal = (run: AbortSignal | undefined, guard: AbortSignal): AbortSignal =>
81
+ run === undefined ? guard : AbortSignal.any([run, guard]);
82
+
70
83
  async function sessionOver(
71
84
  browser: CdpBrowserLike,
72
85
  init: SessionInit,
@@ -99,15 +112,21 @@ async function sessionOver(
99
112
  guard.touch();
100
113
  init.onActivity?.();
101
114
  };
115
+ const signal = withWedgeSignal(init.signal, guard.signal);
102
116
  return {
103
117
  driver: CDP_DRIVER,
118
+ // The exit the browser was launched or attached with, handed back so the run's robots read
119
+ // presents the SAME client identity to the origin the page loads from. `options.proxy` and
120
+ // not `init.proxy`: this is what the launch args and the HTTP leg below actually carry, and a
121
+ // reported exit that nothing dialled would be worse than none.
122
+ ...(options.proxy === undefined ? {} : { proxy: options.proxy }),
104
123
  page: pageOverTarget(target, {
105
124
  clock: init.clock,
106
125
  allowHosts: init.rules.allowHosts,
107
126
  defaultTimeoutMs: init.timeoutMs,
108
127
  secrets: init.secrets,
109
128
  robots: init.robots,
110
- signal: init.signal,
129
+ signal,
111
130
  onActivity,
112
131
  pace: init.pace,
113
132
  }),
@@ -121,7 +140,7 @@ async function sessionOver(
121
140
  session: () => target.session(),
122
141
  robots: init.robots,
123
142
  pace: init.pace,
124
- signal: init.signal,
143
+ signal,
125
144
  onActivity,
126
145
  proxy: options.proxy,
127
146
  }),
package/src/driver.ts CHANGED
@@ -42,6 +42,15 @@ export interface SessionInit {
42
42
 
43
43
  export interface ScrapeSession {
44
44
  readonly driver: string;
45
+ /**
46
+ * The exit this session ACTUALLY dialled — `undefined` for a direct one. Reported because the
47
+ * proxy is resolved inside `open()`, after the robots gate handed to it was built: without this
48
+ * the default `/robots.txt` read leaves from the worker's IP while every page load leaves
49
+ * through the proxy, which is a second client identity presented to the same origin — and an
50
+ * origin reachable only through the proxy then reads as "no robots.txt", which the gate treats
51
+ * as allow-everything. A driver that omits it is asked for its rules directly, as before.
52
+ */
53
+ readonly proxy?: string | undefined;
45
54
  readonly page: ScrapePage;
46
55
  /**
47
56
  * The second transport, bound to the SAME session as the page: the browser's cookies, headers
@@ -75,6 +75,21 @@ export const notActionable = (selector: string, reason: string, waitedMs: number
75
75
  meta: { selector, reason, waitedMs },
76
76
  });
77
77
 
78
+ /**
79
+ * Declared, and structurally unable to fire. `maxDrop` is a fraction of a TRAILING MEDIAN, and the
80
+ * only source of one is a `history:` store — with none, `guardYield` reads an empty array and the
81
+ * `MIN_BASELINE_RUNS` gate is true forever, so the alarm never raises `X_SCRAPE_YIELD_COLLAPSED`
82
+ * on any run. Two halves that must be set together, where setting one alone did nothing and
83
+ * nothing said so.
84
+ */
85
+ export const yieldHistoryMissing = (scrapeName: string): ScrapeError =>
86
+ new ScrapeError({
87
+ code: 'X_SCRAPE_YIELD_HISTORY_MISSING',
88
+ cause: `scrape "${scrapeName}" declares expect.maxDrop with no history: store, so there is no baseline to measure a drop against and the alarm can never fire`,
89
+ fix: `add history: memoryYieldHistory() to scrape("${scrapeName}") for a test, or your own YieldHistory in production — or drop expect.maxDrop and keep expect.minRows, which needs no baseline`,
90
+ meta: { scrape: scrapeName },
91
+ });
92
+
78
93
  export const scrapeTimeout = (what: string, ms: number): ScrapeError =>
79
94
  new ScrapeError({
80
95
  code: 'X_SCRAPE_TIMEOUT',
@@ -203,6 +218,19 @@ export const httpFailed = (url: string, status: number, body: string): ScrapeErr
203
218
  retry: status >= 400 && status < 500 && status !== 429 ? 'terminal' : 'retryable',
204
219
  });
205
220
 
221
+ /**
222
+ * The body outgrew its cap. `AbortSignal.timeout` bounds TIME, not bytes — a 30s stream at 50MB/s
223
+ * is a 1.5GB allocation — so a hostile or merely paginated endpoint could OOM-kill the worker
224
+ * mid-run, taking every other job on it with it.
225
+ */
226
+ export const bodyTooLarge = (url: string, readBytes: number, maxBytes: number): ScrapeError =>
227
+ new ScrapeError({
228
+ code: 'X_SCRAPE_BODY_TOO_LARGE',
229
+ cause: `${url} sent more than ${String(maxBytes)} bytes (${String(readBytes)} read before the read was cancelled)`,
230
+ fix: `raise the cap on this one call — http.request(url, { maxBytes: ${String(maxBytes * 2)} }) — or ask the endpoint for a page instead of the whole collection`,
231
+ meta: { url, readBytes, maxBytes },
232
+ });
233
+
206
234
  /**
207
235
  * TERMINAL, and the retry table cannot be talked out of it. A site that locks an account after
208
236
  * three wrong attempts turns a retrying framework into the thing that destroys the user's
package/src/errors.ts CHANGED
@@ -25,6 +25,7 @@ export const SCRAPE_OWNED_ERROR_CODES = [
25
25
  'X_SCRAPE_PAGE_CRASHED',
26
26
  'X_SCRAPE_OUTPUT_INVALID',
27
27
  'X_SCRAPE_YIELD_COLLAPSED',
28
+ 'X_SCRAPE_YIELD_HISTORY_MISSING',
28
29
  'X_SCRAPE_DOWNLOAD_TIMEOUT',
29
30
  'X_SCRAPE_ROBOTS_DISALLOWED',
30
31
  'X_SCRAPE_FIXTURE_MISSING',
@@ -33,6 +34,7 @@ export const SCRAPE_OWNED_ERROR_CODES = [
33
34
  'X_SCRAPE_RECOVER_REFUSED',
34
35
  'X_SCRAPE_SECRET_EXPOSED',
35
36
  'X_SCRAPE_HTTP_FAILED',
37
+ 'X_SCRAPE_BODY_TOO_LARGE',
36
38
  'X_SCRAPE_AUTH_FAILED',
37
39
  'X_SCRAPE_SESSION_EXPIRED',
38
40
  'X_SCRAPE_PROMPT_UNANSWERED',
@@ -69,6 +71,8 @@ export const SCRAPE_ERROR_TITLES: Readonly<Record<ScrapeOwnedErrorCode, string>>
69
71
  X_SCRAPE_PAGE_CRASHED: 'the renderer process died',
70
72
  X_SCRAPE_OUTPUT_INVALID: 'the extracted rows do not match the extract schema',
71
73
  X_SCRAPE_YIELD_COLLAPSED: 'the run succeeded and returned far too little',
74
+ X_SCRAPE_YIELD_HISTORY_MISSING:
75
+ 'a maxDrop is declared and no history store can supply its baseline',
72
76
  X_SCRAPE_DOWNLOAD_TIMEOUT: 'the download never landed',
73
77
  X_SCRAPE_ROBOTS_DISALLOWED: 'robots.txt disallows this path',
74
78
  X_SCRAPE_FIXTURE_MISSING: 'the fixture driver has no recording for this request',
@@ -77,6 +81,7 @@ export const SCRAPE_ERROR_TITLES: Readonly<Record<ScrapeOwnedErrorCode, string>>
77
81
  X_SCRAPE_RECOVER_REFUSED: 'the recovery hook declined to recover this failure',
78
82
  X_SCRAPE_SECRET_EXPOSED: 'an artifact would have carried a secret this run typed',
79
83
  X_SCRAPE_HTTP_FAILED: 'the site answered the HTTP leg with a non-2xx status',
84
+ X_SCRAPE_BODY_TOO_LARGE: 'the HTTP response body passed its byte cap',
80
85
  X_SCRAPE_AUTH_FAILED: 'the credentials were rejected',
81
86
  X_SCRAPE_SESSION_EXPIRED: 'the restored session is no longer valid and nothing can renew it',
82
87
  X_SCRAPE_PROMPT_UNANSWERED: 'a login step asked for a code and nothing answered',
@@ -126,7 +131,13 @@ export const SCRAPE_ERROR_RETRY = {
126
131
  // half-submitted form is the incident, not the recovery.
127
132
  X_SCRAPE_PAGE_CRASHED: 'terminal',
128
133
  X_SCRAPE_OUTPUT_INVALID: 'terminal',
134
+ // A response size is a property of the endpoint, not of the moment: attempt 2 buffers the same
135
+ // gigabyte and dies the same way. The fix is a number on the request, so a human decides it.
136
+ X_SCRAPE_BODY_TOO_LARGE: 'terminal',
129
137
  X_SCRAPE_YIELD_COLLAPSED: 'terminal',
138
+ // A declaration error, raised by `scrape()` before any attempt exists — there is no run to
139
+ // retry, and the same definition would refuse identically forever.
140
+ X_SCRAPE_YIELD_HISTORY_MISSING: 'terminal',
130
141
  X_SCRAPE_ROBOTS_DISALLOWED: 'terminal',
131
142
  X_SCRAPE_FIXTURE_MISSING: 'terminal',
132
143
  X_SCRAPE_FIXTURE_STALE: 'terminal',
@@ -49,6 +49,18 @@ export interface HtmlTargetInit {
49
49
 
50
50
  const EMPTY: PageRecording = { url: 'about:blank', html: '' };
51
51
 
52
+ /**
53
+ * A recorded map, read by a key that came out of the RECORDING's own markup — a selector, an
54
+ * expression, an `<iframe name>`. Plain indexing walks the prototype chain, so `name="constructor"`
55
+ * resolved to `Object` rather than to `undefined` and `fixtureMissing` never threw; the run then
56
+ * carried a function where an HTML string belongs. Offline driver only, and it is still worth
57
+ * closing: the confusing test failure it produces costs more to read than this line does.
58
+ */
59
+ const recorded = (
60
+ map: Readonly<Record<string, string>> | undefined,
61
+ key: string,
62
+ ): string | undefined => (map !== undefined && Object.hasOwn(map, key) ? map[key] : undefined);
63
+
52
64
  /** Typed text is an overlay keyed by `id`, then `name`, then the selector used to type it. */
53
65
  const keyOf = (selector: string, element: ElementSnapshot | undefined): string => {
54
66
  if (element === undefined) return selector;
@@ -159,10 +171,11 @@ export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
159
171
  return Promise.resolve(page.html);
160
172
  },
161
173
  query,
162
- async click(selector: string, index: number): Promise<void> {
163
- const element = await at(selector, index);
174
+ async click(selector: string): Promise<void> {
175
+ const element = await at(selector, 0);
164
176
  if (element === undefined) throw fixtureMissing(`${page.url} ${selector}`, init.source);
165
- const download = page.downloads?.[selector] ?? page.downloads?.[element.attrs['id'] ?? ''];
177
+ const download =
178
+ recorded(page.downloads, selector) ?? recorded(page.downloads, element.attrs['id'] ?? '');
166
179
  if (download !== undefined) armed = download;
167
180
  const href =
168
181
  element.attrs['data-goto'] ?? (element.tag === 'a' ? element.attrs['href'] : undefined);
@@ -181,12 +194,12 @@ export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
181
194
  },
182
195
  evaluate(expression: string): Promise<unknown> {
183
196
  live();
184
- const recorded = page.evaluate?.[expression];
197
+ const answer = recorded(page.evaluate, expression);
185
198
  // Unrecorded and therefore refused, for the same reason an unrecorded page is: an offline
186
199
  // driver that invented an answer here would make the assertion above it meaningless.
187
- if (recorded === undefined)
200
+ if (answer === undefined)
188
201
  throw fixtureMissing(`${page.url} evaluate(${expression})`, init.source);
189
- return Promise.resolve(JSON.parse(recorded) as unknown);
202
+ return Promise.resolve(JSON.parse(answer) as unknown);
190
203
  },
191
204
  screenshot: (_options: CaptureOptions): Promise<Uint8Array> => Promise.resolve(FAKE_PNG),
192
205
  pdf: (_options: CaptureOptions): Promise<Uint8Array> => Promise.resolve(FAKE_PDF),
@@ -196,18 +209,20 @@ export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
196
209
  session = next;
197
210
  return Promise.resolve();
198
211
  },
199
- download(options: { readonly timeoutMs: number }): Promise<ScrapeDownloadFile> {
212
+ // `async`, so the refusal REJECTS: the method is typed `Promise<ScrapeDownloadFile>` and a
213
+ // synchronous `throw` from one escapes past `download().catch(…)` at every caller.
214
+ async download(options: { readonly timeoutMs: number }): Promise<ScrapeDownloadFile> {
200
215
  if (armed === undefined) throw downloadTimeout(options.timeoutMs, page.url);
201
216
  const { filename, contents } = splitDownload(armed);
202
217
  armed = undefined;
203
- return Promise.resolve({ filename, bytes: new TextEncoder().encode(contents) });
218
+ return { filename, bytes: new TextEncoder().encode(contents) };
204
219
  },
205
220
  async frames(): Promise<readonly FrameRef[]> {
206
221
  const refs: FrameRef[] = [];
207
222
  for (const element of await queryHtml(page.html, 'iframe')) {
208
223
  const name = element.attrs['name'] ?? element.attrs['id'] ?? '';
209
224
  const src = element.attrs['src'] ?? '';
210
- const html = page.frames?.[name] ?? page.frames?.[src];
225
+ const html = recorded(page.frames, name) ?? recorded(page.frames, src);
211
226
  if (html === undefined)
212
227
  throw fixtureMissing(`${page.url} iframe ${name || src}`, init.source);
213
228
  const url = src === '' ? page.url : new URL(src, page.url).toString();
package/src/http.ts CHANGED
@@ -10,25 +10,57 @@
10
10
  // guarantee the page vocabulary makes — and a different exit IP mid-session is exactly what
11
11
  // anti-bot systems look for.
12
12
 
13
+ import { readWithinLimit } from '@ultimat3/core';
13
14
  import type { StandardSchemaV1 } from '@ultimat3/schema';
14
15
  import { parse } from '@ultimat3/schema';
15
16
  import type { ScrapeClock } from './clock';
16
17
  import { cookieHeaderFor } from './cookie-scope';
17
- import { hostBlocked, httpFailed, scrapeTimeout } from './error-throws';
18
+ import { bodyTooLarge, hostBlocked, httpFailed, scrapeTimeout } from './error-throws';
18
19
  import type { InterceptRules } from './intercept';
19
20
  import { interceptVerdict } from './intercept';
20
21
  import type { NetworkRing } from './rings';
21
22
  import type { RobotsGate } from './robots';
22
23
  import type { SessionSnapshot } from './session-state';
23
24
 
25
+ /**
26
+ * Just the call. `typeof fetch` also carries `preconnect`, which no test double and no app wrapper
27
+ * can supply — so an option typed `typeof fetch` was unusable without a double cast, which is
28
+ * exactly what every caller of it had written. The same seam `@ultimat3/cache`, `@ultimat3/auth`
29
+ * and `@ultimat3/mail` already name.
30
+ */
31
+ export type ScrapeFetch = (input: string, init: ScrapeFetchInit) => Promise<Response>;
32
+
33
+ /**
34
+ * `RequestInit` plus the one Bun extension this package sets. Named rather than cast: the DOM's
35
+ * `RequestInit` has no `proxy`, and an `as RequestInit` over the literal silenced the excess-key
36
+ * check for `proxy` AND for every neighbouring key it was standing next to.
37
+ */
38
+ export interface ScrapeFetchInit extends RequestInit {
39
+ /** The session's exit. A different exit IP mid-session is a different client to an anti-bot. */
40
+ readonly proxy?: string | undefined;
41
+ }
42
+
24
43
  export interface HttpRequestInit {
25
44
  readonly method?: string | undefined;
26
45
  readonly headers?: Readonly<Record<string, string>> | undefined;
27
46
  readonly body?: string | undefined;
28
47
  /** Milliseconds. Falls back to the session's own default. */
29
48
  readonly timeout?: number | undefined;
49
+ /**
50
+ * Response-body ceiling in bytes. Falls back to `DEFAULT_HTTP_MAX_BYTES`. Never absent: a
51
+ * deadline bounds time and a scraped endpoint is somebody else's, so the only thing standing
52
+ * between a hostile stream and the worker's heap is a number.
53
+ */
54
+ readonly maxBytes?: number | undefined;
30
55
  }
31
56
 
57
+ /**
58
+ * Generous for the JSON endpoint behind a paginated page — the reason this transport exists — and
59
+ * far under what OOM-kills a worker. Raised per call with `{ maxBytes }`, never globally: a run
60
+ * that genuinely pulls a large export says so at the call site that pulls it.
61
+ */
62
+ export const DEFAULT_HTTP_MAX_BYTES = 32 * 1024 * 1024;
63
+
32
64
  export interface ScrapeResponse {
33
65
  readonly url: string;
34
66
  readonly status: number;
@@ -66,13 +98,26 @@ export interface HttpTransportInit {
66
98
  readonly onActivity?: (() => void) | undefined;
67
99
  /** The SAME proxy the browser dialled through. A different exit IP is a different client. */
68
100
  readonly proxy?: string | undefined;
69
- readonly fetch?: typeof fetch | undefined;
101
+ readonly fetch?: ScrapeFetch | undefined;
70
102
  }
71
103
 
104
+ /**
105
+ * Response headers as data, on a NULL prototype and written with `defineProperty`.
106
+ *
107
+ * The header set on a scraping leg is entirely the site's. `out[key] = value` on a plain object
108
+ * DROPS `__proto__` — a legal HTTP field-name token — because the setter it hits refuses a string
109
+ * and files no own key, and it leaves `headers['toString']` answering a function the site never
110
+ * sent. Both make `Readonly<Record<string, string>>` a lie the caller cannot see through.
111
+ */
72
112
  const headerRecord = (headers: Headers): Record<string, string> => {
73
- const out: Record<string, string> = {};
113
+ const out = Object.create(null) as Record<string, string>;
74
114
  headers.forEach((value, key) => {
75
- out[key] = value;
115
+ Object.defineProperty(out, key, {
116
+ value,
117
+ enumerable: true,
118
+ writable: true,
119
+ configurable: true,
120
+ });
76
121
  });
77
122
  return out;
78
123
  };
@@ -105,7 +150,7 @@ export function responseOver(
105
150
  * robots rule, and neither is re-implemented for the second leg.
106
151
  */
107
152
  export function httpOverFetch(init: HttpTransportInit): ScrapeHttp {
108
- const call = init.fetch ?? fetch;
153
+ const call: ScrapeFetch = init.fetch ?? fetch;
109
154
  return {
110
155
  async request(url: string, request: HttpRequestInit = {}): Promise<ScrapeResponse> {
111
156
  init.onActivity?.();
@@ -135,7 +180,7 @@ export function httpOverFetch(init: HttpTransportInit): ScrapeHttp {
135
180
  ...(request.body === undefined ? {} : { body: request.body }),
136
181
  signal: AbortSignal.any(signals),
137
182
  ...(init.proxy === undefined ? {} : { proxy: init.proxy }),
138
- } as RequestInit);
183
+ });
139
184
  init.network.push({
140
185
  method: request.method ?? 'GET',
141
186
  url,
@@ -143,7 +188,13 @@ export function httpOverFetch(init: HttpTransportInit): ScrapeHttp {
143
188
  resourceType: 'fetch',
144
189
  at: init.clock.now().getTime(),
145
190
  });
146
- const body = await response.text();
191
+ // Counted as it arrives rather than `.text()`, which materialises first and checks never:
192
+ // a 30s stream at 50MB/s is a 1.5GB allocation the worker does not get back, and it takes
193
+ // every other job on that worker with it. The same read `robots-fetch.ts` performs.
194
+ const maxBytes = request.maxBytes ?? DEFAULT_HTTP_MAX_BYTES;
195
+ const capped = await readWithinLimit(response.body, maxBytes);
196
+ if ('over' in capped) throw bodyTooLarge(url, capped.over, maxBytes);
197
+ const body = new TextDecoder().decode(capped.bytes);
147
198
  return responseOver(url, response.status, headerRecord(response.headers), () =>
148
199
  Promise.resolve(body),
149
200
  );
package/src/index.ts CHANGED
@@ -5,7 +5,12 @@
5
5
  export type { ActionabilityState, ActionabilityWait } from './actionability';
6
6
  export { actionabilityProblem, awaitActionable, DEFAULT_POLL_MS, isStable } from './actionability';
7
7
  export type { ArtifactRef, ArtifactWriter, ArtifactWriterInit } from './artifacts';
8
- export { contentTypeFor, createArtifactWriter, DEFAULT_ARTIFACT_PREFIX } from './artifacts';
8
+ export {
9
+ contentTypeFor,
10
+ createArtifactWriter,
11
+ DEFAULT_ARTIFACT_PREFIX,
12
+ DEFAULT_CONTENT_TYPE,
13
+ } from './artifacts';
9
14
  export type {
10
15
  AuthContext,
11
16
  PromptHandler,
@@ -42,6 +47,7 @@ export { FIXTURE_DRIVER, fixtureBrowser, recordingFilename } from './driver-fixt
42
47
  export {
43
48
  authFailed,
44
49
  blocked,
50
+ bodyTooLarge,
45
51
  browserUnreachable,
46
52
  cdpAttachFailed,
47
53
  downloadTimeout,
@@ -97,7 +103,7 @@ export { markupRequests } from './html-requests';
97
103
  export type { HtmlTargetInit, RecordingLookup } from './html-target';
98
104
  export { htmlTarget } from './html-target';
99
105
  export type { HttpRequestInit, HttpTransportInit, ScrapeHttp, ScrapeResponse } from './http';
100
- export { httpOverFetch, responseOver } from './http';
106
+ export { DEFAULT_HTTP_MAX_BYTES, httpOverFetch, responseOver } from './http';
101
107
  export type { HttpRecordingLookup, RecordedHttpInit } from './http-recorded';
102
108
  export { httpRecordingFilename, httpRecordingsOf, recordedHttp } from './http-recorded';
103
109
  export type { InterceptRules, InterceptVerdict } from './intercept';
@@ -137,6 +143,12 @@ export type {
137
143
  export { createRing, DEFAULT_RING_CAPACITY, RESOURCE_TYPES } from './rings';
138
144
  export type { RobotsFetch, RobotsGate, RobotsGateInit, RobotsPolicy, RobotsRules } from './robots';
139
145
  export { createRobotsGate, DEFAULT_ROBOTS_AGENT, parseRobots, robotsAllows } from './robots';
146
+ export type { RobotsFetchInit } from './robots-fetch';
147
+ export {
148
+ DEFAULT_ROBOTS_MAX_BYTES,
149
+ DEFAULT_ROBOTS_TIMEOUT_MS,
150
+ robotsFetcher,
151
+ } from './robots-fetch';
140
152
  export type {
141
153
  ScrapeArtifacts,
142
154
  ScrapeDefinition,
@@ -105,7 +105,7 @@ function frameOver(
105
105
  waitFor: (selector, options) => wait(selector, options, 'actionable'),
106
106
  async click(selector, options): Promise<void> {
107
107
  await wait(selector, options, 'actionable');
108
- await (await resolve()).click(selector, 0);
108
+ await (await resolve()).click(selector);
109
109
  },
110
110
  async type(selector, text, options): Promise<void> {
111
111
  await wait(selector, options, 'actionable');
@@ -187,8 +187,7 @@ export function pageOverTarget(target: ScrapeTarget, ctx: PageContext): ScrapePa
187
187
  // in object storage, forever. Refused rather than masked — a mask over pixels is a guess
188
188
  // about layout, and `page.html()` already gives a redacted artifact that is exact.
189
189
  if (state.tainted) throw secretExposed(kind, target.url());
190
- const timeoutMs = options?.timeout ?? ctx.defaultTimeoutMs;
191
- const request = { fullPage: options?.fullPage, timeoutMs };
190
+ const request = { fullPage: options?.fullPage };
192
191
  return kind === 'screenshot' ? target.screenshot(request) : target.pdf(request);
193
192
  };
194
193
  return {
@@ -204,8 +203,12 @@ export function pageOverTarget(target: ScrapeTarget, ctx: PageContext): ScrapePa
204
203
  },
205
204
  screenshot: (options) => capture('screenshot', options),
206
205
  pdf: (options) => capture('pdf', options),
207
- download: (options?: DownloadRequest): Promise<ScrapeDownloadFile> =>
208
- target.download({ timeoutMs: options?.timeout ?? ctx.defaultTimeoutMs }),
206
+ // `async`, and that is the whole point of the keyword here: `ScrapeTarget` is the seam a third
207
+ // party implements, and a driver that THROWS from its promise-typed `download()` would escape
208
+ // past this page's caller `.catch()` if the forward were a bare arrow.
209
+ async download(options?: DownloadRequest): Promise<ScrapeDownloadFile> {
210
+ return await target.download({ timeoutMs: options?.timeout ?? ctx.defaultTimeoutMs });
211
+ },
209
212
  cookies: (): Promise<readonly ScrapeCookie[]> => target.cookies(),
210
213
  session: () => target.session(),
211
214
  console: () => target.console.entries(),
package/src/page.ts CHANGED
@@ -64,9 +64,9 @@ export interface ScrapeFrame {
64
64
  frame(nameOrSelector: string): ScrapeFrame;
65
65
  }
66
66
 
67
+ /** `fullPage` only. `timeout` is gone with the port's — see `CaptureOptions` for why. */
67
68
  export interface CaptureRequest {
68
69
  readonly fullPage?: boolean | undefined;
69
- readonly timeout?: number | undefined;
70
70
  }
71
71
 
72
72
  export interface DownloadRequest {
@@ -0,0 +1,82 @@
1
+ // The ONE `/robots.txt` read the gate performs when the caller injects no `fetchText`.
2
+ //
3
+ // It exists as its own file because the production default was the only network call in this
4
+ // package with no deadline, no size cap and no proxy — and `scrape-run.ts` builds the gate with no
5
+ // `fetchText`, so production always took it. Every existing gate test injected one, which is how a
6
+ // read that could park a run forever stayed green.
7
+ //
8
+ // The exit is a RESOLVER, not a string: `scrape-run.ts` builds this gate as an argument to
9
+ // `driver.open()`, and the proxy is a driver option the session only reports on the way back out.
10
+
11
+ import { readWithinLimit } from '@ultimat3/core';
12
+ import type { ScrapeFetch } from './http';
13
+ import type { RobotsFetch } from './robots';
14
+
15
+ /**
16
+ * A deadline is applied ALWAYS, proxy or no proxy, session or no session: the failure it prevents
17
+ * is a hung origin whose cached promise then parks every later navigation to that origin, past
18
+ * `ctx.signal`, the watchdog and the job timeout. Ten seconds is long for a static text file and
19
+ * short against a slow-loris.
20
+ */
21
+ export const DEFAULT_ROBOTS_TIMEOUT_MS = 10_000;
22
+
23
+ /** Google's own documented ceiling for the file, and generous for a list of path prefixes. */
24
+ export const DEFAULT_ROBOTS_MAX_BYTES = 500 * 1024;
25
+
26
+ export interface RobotsFetchInit {
27
+ /** Per-read wall clock. Defaults to `DEFAULT_ROBOTS_TIMEOUT_MS`. */
28
+ readonly timeoutMs?: number | undefined;
29
+ /** The run's cancellation, when there is one. Composed with the deadline, never replacing it. */
30
+ readonly signal?: AbortSignal | undefined;
31
+ /**
32
+ * The SAME proxy the browser dialled through, when the session has one — asked PER READ, never
33
+ * captured. A resolver rather than a string because construction order forbids the string: the
34
+ * gate is an argument to `driver.open()` and the proxy is a driver option resolved inside it,
35
+ * so a value passed here could only ever be the one nobody has yet. That is how the robots read
36
+ * came to exit from the worker's IP while every page load exited through the proxy — and how an
37
+ * origin reachable ONLY through the proxy read as "no robots.txt", which is allow-everything.
38
+ *
39
+ * Optional by design: proxies are an opt-in leg, and an origin reachable directly must still be
40
+ * asked for its rules.
41
+ */
42
+ readonly proxy?: (() => string | undefined) | undefined;
43
+ readonly maxBytes?: number | undefined;
44
+ /**
45
+ * The platform `fetch`, injectable so the default path itself is testable. `ScrapeFetch` and not
46
+ * `typeof fetch`: the latter also carries `preconnect`, so nothing a caller can write satisfies
47
+ * it and the option was reachable only through a cast.
48
+ */
49
+ readonly fetch?: ScrapeFetch | undefined;
50
+ }
51
+
52
+ /**
53
+ * Reads `robotsUrl`, or answers `undefined` — which the gate reads as "no restrictions", the
54
+ * standard's own answer for a file it cannot obtain. A deadline that fires, a body past the cap
55
+ * and a 404 are all the same answer on purpose: none of them is evidence of a rule.
56
+ */
57
+ export function robotsFetcher(init: RobotsFetchInit = {}): RobotsFetch {
58
+ const call: ScrapeFetch = init.fetch ?? fetch;
59
+ const limit = init.maxBytes ?? DEFAULT_ROBOTS_MAX_BYTES;
60
+ return async (robotsUrl: string): Promise<string | undefined> => {
61
+ // Armed per read, not per gate: the gate is long-lived and reads once per origin, so a
62
+ // deadline created alongside it would already have expired by the second origin.
63
+ const deadline = AbortSignal.timeout(init.timeoutMs ?? DEFAULT_ROBOTS_TIMEOUT_MS);
64
+ const signal = init.signal === undefined ? deadline : AbortSignal.any([deadline, init.signal]);
65
+ // Resolved here, at the read, because the session that owns the exit did not exist when this
66
+ // fetcher was built. An empty string is not an exit and is dropped with the absent one.
67
+ const proxy = init.proxy?.();
68
+ try {
69
+ const response = await call(robotsUrl, {
70
+ signal,
71
+ ...(proxy === undefined || proxy === '' ? {} : { proxy }),
72
+ });
73
+ if (!response.ok) return undefined;
74
+ // Counted as it arrives rather than `.text()`, which materialises the whole body first: a
75
+ // multi-gigabyte robots.txt is a heap the worker never gets back.
76
+ const capped = await readWithinLimit(response.body, limit);
77
+ return 'over' in capped ? undefined : new TextDecoder().decode(capped.bytes);
78
+ } catch {
79
+ return undefined;
80
+ }
81
+ };
82
+ }
package/src/robots.ts CHANGED
@@ -6,6 +6,8 @@
6
6
  // There is no boolean, because `robots: false` is a decision with no author.
7
7
 
8
8
  import { robotsDisallowed } from './error-throws';
9
+ import type { RobotsFetchInit } from './robots-fetch';
10
+ import { robotsFetcher } from './robots-fetch';
9
11
 
10
12
  export type RobotsPolicy = 'obey' | { readonly ignore: string };
11
13
 
@@ -56,14 +58,47 @@ export function parseRobots(text: string, agent: string): RobotsRules {
56
58
  return { rules: groups.get(wanted) ?? groups.get('*') ?? [] };
57
59
  }
58
60
 
59
- const escaped = (literal: string): string => literal.replaceAll(/[.*+?^${}()|[\]\\]/g, '\\$&');
60
-
61
- /** `/private/*.pdf$` — the two wildcards robots.txt defines, and no others. */
61
+ /**
62
+ * `/private/*.pdf$` — the two wildcards robots.txt defines, and no others, matched by WALKING the
63
+ * pattern rather than by compiling one.
64
+ *
65
+ * The rule text is remote: robots.txt belongs to the site being scraped, and `robotsAllows` runs on
66
+ * every navigation and every HTTP-leg request, synchronously, on the worker's only thread. A regex
67
+ * built as `body.split('*').join('.*')` backtracks catastrophically on a non-matching path — a rule
68
+ * with 24 wildcards did not return inside 60s, past `ctx.signal`, the watchdog and the job timeout,
69
+ * all of which are downstream of a `return` that never happens. This walk is O(pattern × path) with
70
+ * no backtracking beyond the LAST star, so the class is removed rather than bounded.
71
+ */
62
72
  const patternMatches = (pattern: string, path: string): boolean => {
63
73
  const anchored = pattern.endsWith('$');
64
- const body = anchored ? pattern.slice(0, -1) : pattern;
65
- const source = body.split('*').map(escaped).join('.*');
66
- return new RegExp(`^${source}${anchored ? '$' : ''}`).test(path);
74
+ // Unanchored means "matches a PREFIX of the path", which is the same statement as a full match
75
+ // against the pattern with one more `*` on the end — one code path instead of two.
76
+ const body = anchored ? pattern.slice(0, -1) : `${pattern}*`;
77
+ let p = 0;
78
+ let s = 0;
79
+ let star = -1;
80
+ let resume = 0;
81
+ while (s < path.length) {
82
+ if (p < body.length && body[p] === '*') {
83
+ star = p;
84
+ p += 1;
85
+ resume = s;
86
+ continue;
87
+ }
88
+ if (p < body.length && body[p] === path[s]) {
89
+ p += 1;
90
+ s += 1;
91
+ continue;
92
+ }
93
+ if (star === -1) return false;
94
+ // Only the most recent star is ever retried, which is what keeps this linear per star instead
95
+ // of exponential across all of them.
96
+ p = star + 1;
97
+ resume += 1;
98
+ s = resume;
99
+ }
100
+ while (p < body.length && body[p] === '*') p += 1;
101
+ return p === body.length;
67
102
  };
68
103
 
69
104
  /**
@@ -93,14 +128,10 @@ export interface RobotsGate {
93
128
 
94
129
  export type RobotsFetch = (robotsUrl: string) => Promise<string | undefined>;
95
130
 
96
- const fetchRobots: RobotsFetch = async (robotsUrl) => {
97
- const response = await fetch(robotsUrl);
98
- return response.ok ? await response.text() : undefined;
99
- };
100
-
101
- export interface RobotsGateInit {
131
+ export interface RobotsGateInit extends RobotsFetchInit {
102
132
  readonly policy: RobotsPolicy;
103
133
  readonly agent?: string | undefined;
134
+ /** A caller-supplied read. With none, `robotsFetcher` builds the deadlined, capped default. */
104
135
  readonly fetchText?: RobotsFetch | undefined;
105
136
  }
106
137
 
@@ -116,7 +147,7 @@ export function createRobotsGate(init: RobotsGateInit): RobotsGate {
116
147
  return { assertAllowed: () => Promise.resolve(), ignoredBecause: init.policy.ignore };
117
148
  }
118
149
  const agent = init.agent ?? DEFAULT_ROBOTS_AGENT;
119
- const read = init.fetchText ?? fetchRobots;
150
+ const read = init.fetchText ?? robotsFetcher(init);
120
151
  const cache = new Map<string, Promise<RobotsRules>>();
121
152
  return {
122
153
  ignoredBecause: undefined,
package/src/scrape-run.ts CHANGED
@@ -94,18 +94,36 @@ export async function runScrape<I, Row>(
94
94
  // Read BEFORE the browser opens: a refused credential must not reach a login form again, and
95
95
  // opening a session first would already have spent an identity on a run that cannot succeed.
96
96
  const restored = await restorableSession(plan);
97
+ const pageTimeoutMs = toMillis(definition.pageTimeout, DEFAULT_PAGE_TIMEOUT_MS);
98
+ // The exit the session dials, readable only AFTER `driver.open()` — the proxy is a driver
99
+ // option and the gate below is an argument to `open()`, so the gate asks for it per read
100
+ // instead of being handed a value that cannot exist yet. Every read happens during a
101
+ // navigation, which is after this is assigned.
102
+ let sessionProxy: string | undefined;
97
103
  const session = await driver.open({
98
104
  name: definition.name,
99
105
  rules,
100
106
  clock,
101
- timeoutMs: toMillis(definition.pageTimeout, DEFAULT_PAGE_TIMEOUT_MS),
107
+ timeoutMs: pageTimeoutMs,
102
108
  secrets,
103
- robots: createRobotsGate({ policy: definition.robots ?? 'obey' }),
109
+ // The gate reads `/robots.txt` over the network, so it gets the run's deadline, the run's
110
+ // cancellation and the run's exit, like every other call this package makes. Without the
111
+ // first two a hung origin parks every later navigation to it on one cached promise,
112
+ // unreachable by `ctx.signal`; without the third the read leaves from a different IP than
113
+ // every page load, and an origin reachable only through the proxy answers nothing — which
114
+ // this gate reads as "no restrictions".
115
+ robots: createRobotsGate({
116
+ policy: definition.robots ?? 'obey',
117
+ timeoutMs: pageTimeoutMs,
118
+ signal: args.ctx.signal,
119
+ proxy: () => sessionProxy,
120
+ }),
104
121
  signal: args.ctx.signal,
105
122
  restore: restored,
106
123
  pace: (signal) => pace(signal),
107
124
  watchdog: definition.watchdog,
108
125
  });
126
+ sessionProxy = session.proxy;
109
127
 
110
128
  try {
111
129
  if (definition.auth !== undefined) {
@@ -155,8 +173,11 @@ export async function runScrape<I, Row>(
155
173
  };
156
174
  } catch (thrown) {
157
175
  logger.error('scrape.failed', { code: errorCode(thrown) });
158
- if (errorCode(thrown) === 'X_SCRAPE_AUTH_FAILED') await markRefused(plan);
159
- else if (burnsSession(thrown)) await burnSession(plan);
176
+ if (errorCode(thrown) === 'X_SCRAPE_AUTH_FAILED') {
177
+ await recordSessionOutcome('session.refuse', logger, () => markRefused(plan));
178
+ } else if (burnsSession(thrown)) {
179
+ await recordSessionOutcome('session.burn', logger, () => burnSession(plan));
180
+ }
160
181
  if (definition.artifacts?.onFailure !== false) await saveFailureArtifact(session, artifact);
161
182
  throw thrown;
162
183
  } finally {
@@ -165,6 +186,29 @@ export async function runScrape<I, Row>(
165
186
  }
166
187
  }
167
188
 
189
+ /**
190
+ * The tombstone or the burn, on the way out — best effort, and it may NEVER replace the failure
191
+ * that caused it. `markRefused` reaches `store.save()` reaches `storage.put()`, so an S3 503 or an
192
+ * `X_STORAGE_PATH_UNSAFE` from a tenant whose key sanitises to nothing used to propagate out of
193
+ * the catch and REPLACE a terminal `X_SCRAPE_AUTH_FAILED` with a retryable one. Attempt 2 then
194
+ * found no tombstone — the save is what failed — and walked the same rejected password back to
195
+ * the login form; attempt 3 locks the account. Same rule as `saveFailureArtifact`, and the same
196
+ * reason: the run's own error is the one the reader needs.
197
+ */
198
+ async function recordSessionOutcome(
199
+ step: 'session.refuse' | 'session.burn',
200
+ logger: ReturnType<typeof scrapeLogger>,
201
+ write: () => Promise<void>,
202
+ ): Promise<void> {
203
+ try {
204
+ await write();
205
+ } catch (thrown) {
206
+ // Logged rather than swallowed silently: the tombstone is missing, so the NEXT attempt will
207
+ // re-probe rather than refuse cheaply, and that is a fact an operator has to be able to see.
208
+ logger.error('scrape.session.write_failed', { step, code: errorCode(thrown) });
209
+ }
210
+ }
211
+
168
212
  /**
169
213
  * The body, with at most ONE recovery pass. `recover` never runs for a failure that must not be
170
214
  * retried — a rejected credential is the case, and asking a model to "fix" a wrong password is
package/src/scrape.ts CHANGED
@@ -18,6 +18,7 @@ import type { ArtifactWriter } from './artifacts';
18
18
  import type { PromptHandler, ScrapeAuth } from './auth';
19
19
  import type { ScrapeClock } from './clock';
20
20
  import type { ScrapeDriver } from './driver';
21
+ import { yieldHistoryMissing } from './error-throws';
21
22
  import type { YieldExpectation, YieldHistory } from './expect';
22
23
  import type { HostRule } from './hosts';
23
24
  import type { ScrapeHttp } from './http';
@@ -137,6 +138,12 @@ export function scrape<I, Row>(definition: ScrapeDefinition<I, Row>): JobHandle<
137
138
  `scrape "${definition.name}" declares rate: ${String(definition.rate)} — a rate is navigations per second, greater than zero`,
138
139
  `set rate: 1 on scrape("${definition.name}"), or leave it out — to go faster raise the number, there is no unpaced mode`,
139
140
  );
141
+ // Two halves that must be set together. `maxDrop` is a fraction of a trailing median and only
142
+ // `history:` can supply one, so declaring it alone is an alarm that cannot fire — refused here,
143
+ // where it is written, rather than discovered as a scrape that never once went red.
144
+ if (definition.expect?.maxDrop !== undefined && definition.history === undefined) {
145
+ throw yieldHistoryMissing(definition.name);
146
+ }
140
147
  return job<I>({
141
148
  name: definition.name,
142
149
  input: definition.input,
package/src/secrets.ts CHANGED
@@ -78,11 +78,25 @@ export function redactSecrets(text: string, secrets: ScrapeSecrets | undefined):
78
78
  return out;
79
79
  }
80
80
 
81
- /** `<input type="password" value="hunter2">` -> `value=""`, whatever the value happened to be. */
81
+ /** Every `<input …>` tag, whole, so the rewrite below never has to reason about attribute order. */
82
+ const INPUT_TAG = /<input\b[^>]*>/gi;
83
+ /** `type=password`, quoted either way or bare. */
84
+ const PASSWORD_TYPE = /\btype\s*=\s*(?:"password"|'password'|password)(?=[\s/>])/i;
85
+ const VALUE_ATTR = /\bvalue\s*=\s*(?:"[^"]*"|'[^']*'|[^\s>]*)/gi;
86
+
87
+ /**
88
+ * `<input type="password" value="hunter2">` -> `value=""`, whatever the value happened to be —
89
+ * and whatever ORDER the site wrote the attributes in.
90
+ *
91
+ * One regex over the whole tag required `type` to precede `value`, so `<input value="hunter2"
92
+ * type="password">` came through untouched. Attribute order is the site's choice, and
93
+ * `saveFailureArtifact` writes `page.html()` to object storage on every failed run: a
94
+ * server-rendered password on a reversed-attribute form was durably persisted. So: match the TAG
95
+ * first, then rewrite `value` inside it.
96
+ */
82
97
  export function blankPasswordFields(html: string): string {
83
- return html.replaceAll(
84
- /(<input\b[^>]*\btype\s*=\s*['"]?password['"]?[^>]*?)\bvalue\s*=\s*(?:"[^"]*"|'[^']*'|[^\s>]*)/gi,
85
- '$1value=""',
98
+ return html.replaceAll(INPUT_TAG, (tag) =>
99
+ PASSWORD_TYPE.test(tag) ? tag.replaceAll(VALUE_ATTR, 'value=""') : tag,
86
100
  );
87
101
  }
88
102
 
@@ -157,11 +157,35 @@ export function storageSessionStore(
157
157
  };
158
158
  }
159
159
 
160
- const isCookie = (value: unknown): value is ScrapeCookie =>
161
- typeof value === 'object' &&
162
- value !== null &&
163
- typeof (value as { name?: unknown }).name === 'string' &&
164
- typeof (value as { value?: unknown }).value === 'string';
160
+ /**
161
+ * A stored cookie is somebody else's JSON. `name` and `value` are what makes it a cookie at all;
162
+ * the four scope fields `ScrapeCookie` REQUIRES are completed here rather than asserted.
163
+ *
164
+ * Asserting them was the bug: this was a `value is ScrapeCookie` predicate that checked two of
165
+ * that type's six required fields, so a stored `{ name, value }` left `parseSessionState` typed as
166
+ * a whole cookie with no `domain` — and `cookieHeaderFor`, a public export, hands it to
167
+ * `cookieDomainMatches`, which calls `.trim()` on it and throws a bare `TypeError`.
168
+ *
169
+ * The defaults are the ones `cookie-scope.ts` already documents. An empty `domain` matches NO
170
+ * host, which is the point: an unscoped cookie must reach nothing, because the only other way to
171
+ * scope it is to infer the domain from whichever URL is asking, and that is exactly how a
172
+ * `bank.test` session cookie reaches `evilbank.test`. `/` is §5.1.4's reading of an absent path,
173
+ * and an attribute a jar never wrote is `false`.
174
+ */
175
+ const toCookie = (value: unknown): ScrapeCookie | undefined => {
176
+ if (typeof value !== 'object' || value === null) return undefined;
177
+ const entry = value as Partial<ScrapeCookie>;
178
+ if (typeof entry.name !== 'string' || typeof entry.value !== 'string') return undefined;
179
+ return {
180
+ name: entry.name,
181
+ value: entry.value,
182
+ domain: typeof entry.domain === 'string' ? entry.domain : '',
183
+ path: typeof entry.path === 'string' ? entry.path : '/',
184
+ ...(typeof entry.expires === 'number' ? { expires: entry.expires } : {}),
185
+ httpOnly: entry.httpOnly === true,
186
+ secure: entry.secure === true,
187
+ };
188
+ };
165
189
 
166
190
  /** Stored JSON is `unknown`. Read structurally, and answer `undefined` rather than half a session. */
167
191
  export function parseSessionState(raw: unknown, key: string): SessionState | undefined {
@@ -172,7 +196,7 @@ export function parseSessionState(raw: unknown, key: string): SessionState | und
172
196
  key,
173
197
  savedAt: value.savedAt,
174
198
  ...(typeof value.refusedAt === 'string' ? { refusedAt: value.refusedAt } : {}),
175
- cookies: value.cookies.filter((cookie): cookie is ScrapeCookie => isCookie(cookie)),
199
+ cookies: value.cookies.flatMap((cookie: unknown) => toCookie(cookie) ?? []),
176
200
  headers: value.headers ?? {},
177
201
  storage: value.storage ?? {},
178
202
  userAgent: value.userAgent ?? '',
package/src/target.ts CHANGED
@@ -74,9 +74,17 @@ export interface GotoOptions {
74
74
  readonly signal?: AbortSignal | undefined;
75
75
  }
76
76
 
77
+ /**
78
+ * `fullPage` and nothing else. It carried a required `timeoutMs` until 2026-08 that NO driver
79
+ * honoured — `cdp-target.ts` read only `fullPage`, `html-target.ts` ignored the whole object —
80
+ * so `page.screenshot({ timeout })` was a documented deadline that bounded nothing. Deleted
81
+ * rather than implemented: the CDP port's own `screenshot({ fullPage })` has no timeout slot to
82
+ * forward it to, and a deadline enforced in `page-over-target.ts` would have to race
83
+ * `ScrapeClock.sleep`, which under `testClock` resolves on the first microtask and would time
84
+ * out every capture in every test. A driver's own default is the honest bound.
85
+ */
77
86
  export interface CaptureOptions {
78
87
  readonly fullPage?: boolean | undefined;
79
- readonly timeoutMs: number;
80
88
  }
81
89
 
82
90
  /**
@@ -93,7 +101,16 @@ export interface ScrapeTarget {
93
101
  /** Serialised HTML of THIS target — the document for a page, the subtree for a frame. */
94
102
  content(): Promise<string>;
95
103
  query(selector: string): Promise<readonly ElementSnapshot[]>;
96
- click(selector: string, index: number): Promise<void>;
104
+ /**
105
+ * Clicks the FIRST match. It took an `index` until 2026-08 that `html-target.ts` honoured and
106
+ * `cdp-target.ts` dropped — its implementations are `click: (selector) => …`, so puppeteer
107
+ * clicked match 0 whatever was asked. They agreed only because `page-over-target.ts`, the sole
108
+ * caller, always passed `0`, and no public vocabulary could set it: `ScrapeFrame.click` takes
109
+ * `(selector, options?: WaitOptions)`. A port member no app can reach and one driver ignores is
110
+ * a divergence waiting to be found by an app, so it is gone. `driver-parity.test.ts` pins that
111
+ * all three drivers click the first match.
112
+ */
113
+ click(selector: string): Promise<void>;
97
114
  /** Appends, exactly as typing does. Clearing first is `fill`'s job at the page level. */
98
115
  type(selector: string, text: string): Promise<void>;
99
116
  clear(selector: string): Promise<void>;