@ultimat3/scraping 13.0.0 → 15.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -7,7 +7,13 @@
7
7
 
8
8
  import type { CaptureClip } from './capture-clip';
9
9
  import type { ScrapeClock } from './clock';
10
- import { browserUnreachable, downloadTimeout, fixtureMissing, fixtureStale } from './error-throws';
10
+ import {
11
+ browserUnreachable,
12
+ downloadTimeout,
13
+ fixtureMissing,
14
+ fixtureStale,
15
+ scrapeNotImplemented,
16
+ } from './error-throws';
11
17
  import { queryHtml } from './html-query';
12
18
  import { markupRequests } from './html-requests';
13
19
  import type { InterceptRules } from './intercept';
@@ -88,6 +94,20 @@ const keyOf = (selector: string, element: ElementSnapshot | undefined): string =
88
94
  return selector;
89
95
  };
90
96
 
97
+ /**
98
+ * ONE browsing context: the page's document, or a frame's. A browser gives every context its own
99
+ * DOM and its own form state, so this target gives every one its own markup and its own overlay.
100
+ *
101
+ * It is the whole fix for the frame half. The frame target used to be `{ ...base }` with `query`
102
+ * overridden, so `type`, `clear` and `select` read the element out of the PARENT's markup and
103
+ * wrote into the PARENT's overlay — a value typed into an iframe'd login form then read back out
104
+ * of `page.values()`, and a `fill` seeded itself from the parent field's value.
105
+ */
106
+ interface OfflineDocument {
107
+ html(): string;
108
+ readonly overlay: Map<string, string>;
109
+ }
110
+
91
111
  export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
92
112
  const consoleRing = createRing<ConsoleLine>();
93
113
  const networkRing = createRing<NetworkEntry>();
@@ -98,6 +118,13 @@ export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
98
118
  // `driver-parity.test.ts` pins the divergence, beside the box/hit-target one it already carries.
99
119
  const pageErrorRing = createRing<PageError>();
100
120
  const overlay = new Map<string, string>();
121
+ /**
122
+ * One overlay per FRAME, held here rather than in the frame target, because `frames()` builds a
123
+ * fresh target on every call — `page.frame(name)` re-resolves per operation, by design — so an
124
+ * overlay owned by the target would be discarded between the `clear` and the `type` a single
125
+ * `fill` performs. Keyed by name and URL together: two frames may share either one alone.
126
+ */
127
+ const frameOverlays = new Map<string, Map<string, string>>();
101
128
  let page: PageRecording = init.start ?? EMPTY;
102
129
  let armed: string | undefined;
103
130
  let closed = false;
@@ -110,19 +137,57 @@ export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
110
137
  if (closed) throw browserUnreachable(init.driver, 'the offline target is already closed');
111
138
  };
112
139
 
113
- const withOverlay = (selector: string, elements: readonly ElementSnapshot[]): ElementSnapshot[] =>
140
+ const withOverlay = (
141
+ document: OfflineDocument,
142
+ selector: string,
143
+ elements: readonly ElementSnapshot[],
144
+ ): ElementSnapshot[] =>
114
145
  elements.map((element) => {
115
- const typed = overlay.get(keyOf(selector, element));
146
+ const typed = document.overlay.get(keyOf(selector, element));
116
147
  return typed === undefined ? element : { ...element, value: typed };
117
148
  });
118
149
 
119
- const query = async (selector: string): Promise<readonly ElementSnapshot[]> => {
150
+ const queryIn = async (
151
+ document: OfflineDocument,
152
+ selector: string,
153
+ ): Promise<readonly ElementSnapshot[]> => {
120
154
  live();
121
- return withOverlay(selector, await queryHtml(page.html, selector));
155
+ return withOverlay(document, selector, await queryHtml(document.html(), selector));
156
+ };
157
+
158
+ const atIn = (
159
+ document: OfflineDocument,
160
+ selector: string,
161
+ index: number,
162
+ ): Promise<ElementSnapshot | undefined> =>
163
+ queryIn(document, selector).then((elements) => elements[index]);
164
+
165
+ /** Appends, exactly as typing does — the port's contract. `fill` is a `clear` and then this. */
166
+ const typeIn = async (
167
+ document: OfflineDocument,
168
+ selector: string,
169
+ text: string,
170
+ ): Promise<void> => {
171
+ const element = await atIn(document, selector, 0);
172
+ const key = keyOf(selector, element);
173
+ document.overlay.set(key, `${document.overlay.get(key) ?? element?.value ?? ''}${text}`);
122
174
  };
123
175
 
176
+ const setIn = async (
177
+ document: OfflineDocument,
178
+ selector: string,
179
+ value: string,
180
+ ): Promise<void> => {
181
+ document.overlay.set(keyOf(selector, await atIn(document, selector, 0)), value);
182
+ };
183
+
184
+ const pageDocument: OfflineDocument = { html: () => page.html, overlay };
185
+
186
+ const query = (selector: string): Promise<readonly ElementSnapshot[]> =>
187
+ queryIn(pageDocument, selector);
188
+
124
189
  const at = (selector: string, index: number): Promise<ElementSnapshot | undefined> =>
125
- query(selector).then((elements) => elements[index]);
190
+ atIn(pageDocument, selector, index);
126
191
 
127
192
  /** Interception, offline: every request the markup would make, judged by the same rule. */
128
193
  const intercept = async (recording: PageRecording): Promise<void> => {
@@ -152,6 +217,9 @@ export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
152
217
  }
153
218
  page = found;
154
219
  overlay.clear();
220
+ // A navigation is a new document tree, frames included: a value typed into the old page's
221
+ // frame must not answer a query on the new one's.
222
+ frameOverlays.clear();
155
223
  armed = undefined;
156
224
  networkRing.push({
157
225
  method: 'GET',
@@ -175,13 +243,46 @@ export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
175
243
  await load(absolute);
176
244
  };
177
245
 
178
- const frameTarget = (html: string, url: string): ScrapeTarget => ({
179
- ...base,
180
- url: () => url,
181
- content: () => Promise.resolve(html),
182
- query: async (selector) => withOverlay(selector, await queryHtml(html, selector)),
183
- frames: () => Promise.resolve([]),
184
- });
246
+ const overlayFor = (key: string): Map<string, string> => {
247
+ const found = frameOverlays.get(key);
248
+ if (found !== undefined) return found;
249
+ const created = new Map<string, string>();
250
+ frameOverlays.set(key, created);
251
+ return created;
252
+ };
253
+
254
+ /**
255
+ * A frame target: the parent's, with every verb that touches a DOCUMENT re-pointed at this
256
+ * frame's. `...base` is what makes the spread dangerous — a verb nobody overrides silently acts
257
+ * on the parent — so the act-verbs are listed here even where the body is one line.
258
+ *
259
+ * `click` navigates NOTHING, and that is a real limit rather than an oversight: a
260
+ * `PageRecording.frames` entry is one static document, so there is no second frame document for
261
+ * a click to land on. What it must never do is navigate the PARENT, which is what inheriting
262
+ * `base.click` did — `driver-parity-frames.test.ts` pins that on all three drivers. `evaluate` and
263
+ * `download` stay the parent's: the recording format keys evaluations and downloads per PAGE,
264
+ * so a frame has no map of its own to read.
265
+ */
266
+ const frameTarget = (html: string, url: string, key: string): ScrapeTarget => {
267
+ const document: OfflineDocument = { html: () => html, overlay: overlayFor(key) };
268
+ return {
269
+ ...base,
270
+ url: () => url,
271
+ content: () => Promise.resolve(html),
272
+ query: (selector) => queryIn(document, selector),
273
+ async click(selector: string): Promise<void> {
274
+ const element = await atIn(document, selector, 0);
275
+ if (element === undefined) throw fixtureMissing(`${url} ${selector}`, init.source);
276
+ const download =
277
+ recorded(page.downloads, selector) ?? recorded(page.downloads, element.attrs['id'] ?? '');
278
+ if (download !== undefined) armed = download;
279
+ },
280
+ type: (selector, text) => typeIn(document, selector, text),
281
+ clear: (selector) => setIn(document, selector, ''),
282
+ select: (selector, values) => setIn(document, selector, values[0] ?? ''),
283
+ frames: () => Promise.resolve([]),
284
+ };
285
+ };
185
286
 
186
287
  const base: ScrapeTarget = {
187
288
  driver: init.driver,
@@ -205,17 +306,10 @@ export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
205
306
  element.attrs['data-goto'] ?? (element.tag === 'a' ? element.attrs['href'] : undefined);
206
307
  if (href !== undefined && href !== '') await navigate(href);
207
308
  },
208
- async type(selector: string, text: string): Promise<void> {
209
- const element = await at(selector, 0);
210
- const key = keyOf(selector, element);
211
- overlay.set(key, `${overlay.get(key) ?? element?.value ?? ''}${text}`);
212
- },
213
- async clear(selector: string): Promise<void> {
214
- overlay.set(keyOf(selector, await at(selector, 0)), '');
215
- },
216
- async select(selector: string, values: readonly string[]): Promise<void> {
217
- overlay.set(keyOf(selector, await at(selector, 0)), values[0] ?? '');
218
- },
309
+ type: (selector: string, text: string): Promise<void> => typeIn(pageDocument, selector, text),
310
+ clear: (selector: string): Promise<void> => setIn(pageDocument, selector, ''),
311
+ select: (selector: string, values: readonly string[]): Promise<void> =>
312
+ setIn(pageDocument, selector, values[0] ?? ''),
219
313
  evaluate(expression: string): Promise<unknown> {
220
314
  live();
221
315
  const answer = recorded(page.evaluate, expression);
@@ -225,6 +319,20 @@ export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
225
319
  throw fixtureMissing(`${page.url} evaluate(${expression})`, init.source);
226
320
  return Promise.resolve(JSON.parse(answer) as unknown);
227
321
  },
322
+ /**
323
+ * REFUSED, and `async` so it REJECTS. There is no browser here and no service worker, so
324
+ * there is no network to cut — and a resolved promise would let "a like taken offline is
325
+ * queued" pass against an app that was online for the whole test. That is the exact shape of
326
+ * lie `packages/testing`'s `fetch` patch already tells about a browser's own requests.
327
+ */
328
+ // `async`, so the refusal REJECTS: the method is typed `Promise<void>` and a synchronous
329
+ // `throw` from one jumps straight over `page.offline(true).catch(…)` at every caller.
330
+ async setOfflineMode(_enabled: boolean): Promise<void> {
331
+ throw scrapeNotImplemented(
332
+ `setOfflineMode() on the ${init.driver} driver`,
333
+ 'run this assertion on localBrowser()/remoteBrowser(), whose setOfflineMode() reaches a real browser — an offline driver has no network to cut, so it cannot prove an offline behaviour',
334
+ );
335
+ },
228
336
  screenshot: (options: CaptureOptions): Promise<Uint8Array> =>
229
337
  Promise.resolve(options.clip === undefined ? FAKE_PNG : clippedPng(options.clip)),
230
338
  pdf: (_options: CaptureOptions): Promise<Uint8Array> => Promise.resolve(FAKE_PDF),
@@ -255,7 +363,7 @@ export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
255
363
  name,
256
364
  url,
257
365
  selector: element.attrs['id'] === undefined ? undefined : `#${element.attrs['id']}`,
258
- target: frameTarget(html, url),
366
+ target: frameTarget(html, url, `${name}\u0000${url}`),
259
367
  });
260
368
  }
261
369
  return refs;
@@ -14,6 +14,7 @@ import { interceptVerdict } from './intercept';
14
14
  import type { HttpRecording } from './recording';
15
15
  import type { NetworkRing } from './rings';
16
16
  import type { RobotsGate } from './robots';
17
+ import type { ScrapeSecrets } from './secrets';
17
18
 
18
19
  export type HttpRecordingLookup = (
19
20
  method: string,
@@ -33,6 +34,8 @@ export interface RecordedHttpInit {
33
34
  */
34
35
  readonly robots?: RobotsGate | undefined;
35
36
  readonly maxAgeMs?: number | undefined;
37
+ /** The SAME bag the live leg holds, so a recorded 4xx body redacts the way a real one does. */
38
+ readonly secrets?: ScrapeSecrets | undefined;
36
39
  }
37
40
 
38
41
  /** `GET https://api.example.com/v1/orders?page=2` -> `http-get-api-example-com-v1-orders-page-2`. */
@@ -67,8 +70,12 @@ export function recordedHttp(init: RecordedHttpInit): ScrapeHttp {
67
70
  resourceType: 'fetch',
68
71
  at: init.clock.now().getTime(),
69
72
  });
70
- return responseOver(url, found.status, found.headers ?? {}, () =>
71
- Promise.resolve(found.body),
73
+ return responseOver(
74
+ url,
75
+ found.status,
76
+ found.headers ?? {},
77
+ () => Promise.resolve(found.body),
78
+ init.secrets,
72
79
  );
73
80
  },
74
81
  };
package/src/http.ts CHANGED
@@ -20,6 +20,8 @@ import type { InterceptRules } from './intercept';
20
20
  import { interceptVerdict } from './intercept';
21
21
  import type { NetworkRing } from './rings';
22
22
  import type { RobotsGate } from './robots';
23
+ import type { ScrapeSecrets } from './secrets';
24
+ import { redactSecrets } from './secrets';
23
25
  import type { SessionSnapshot } from './session-state';
24
26
 
25
27
  /**
@@ -98,6 +100,14 @@ export interface HttpTransportInit {
98
100
  readonly onActivity?: (() => void) | undefined;
99
101
  /** The SAME proxy the browser dialled through. A different exit IP is a different client. */
100
102
  readonly proxy?: string | undefined;
103
+ /**
104
+ * The run's secret bag, for the ONE thing this leg persists that the site wrote: the first 200
105
+ * bytes of a non-2xx body, in `X_SCRAPE_HTTP_FAILED`'s cause. A login endpoint that echoes the
106
+ * submitted credential in its 4xx body put a password in an `UltimateError.message`, which the
107
+ * job driver writes to the dead-letter row and `x jobs show` prints. This field did not exist
108
+ * until 2026-08-24, so nothing on this leg COULD redact.
109
+ */
110
+ readonly secrets?: ScrapeSecrets | undefined;
101
111
  readonly fetch?: ScrapeFetch | undefined;
102
112
  }
103
113
 
@@ -122,11 +132,18 @@ const headerRecord = (headers: Headers): Record<string, string> => {
122
132
  return out;
123
133
  };
124
134
 
135
+ /**
136
+ * `secrets` is optional and last so every existing caller compiles — but a caller that HAS a bag
137
+ * and omits it is a caller whose refusal quotes the site verbatim, which is exactly the defect.
138
+ * Both transports pass it: `httpOverFetch` from the driver's `SessionInit`, `recordedHttp` from
139
+ * the offline session's, so a fixture proves the redaction the live leg performs.
140
+ */
125
141
  export function responseOver(
126
142
  url: string,
127
143
  status: number,
128
144
  headers: Readonly<Record<string, string>>,
129
145
  body: () => Promise<string>,
146
+ secrets?: ScrapeSecrets | undefined,
130
147
  ): ScrapeResponse {
131
148
  const ok = status >= 200 && status < 300;
132
149
  const text = body;
@@ -138,7 +155,9 @@ export function responseOver(
138
155
  text,
139
156
  json: async (): Promise<unknown> => JSON.parse(await text()) as unknown,
140
157
  async parse<T>(schema: StandardSchemaV1<unknown, T>): Promise<T> {
141
- if (!ok) throw httpFailed(url, status, (await text()).slice(0, 200));
158
+ // Redacted BEFORE the slice, never after: cutting at 200 bytes can leave half a secret,
159
+ // and half a password is still half a password in a durable row.
160
+ if (!ok) throw httpFailed(url, status, redactSecrets(await text(), secrets).slice(0, 200));
142
161
  return parse(schema, JSON.parse(await text()) as unknown);
143
162
  },
144
163
  };
@@ -195,8 +214,12 @@ export function httpOverFetch(init: HttpTransportInit): ScrapeHttp {
195
214
  const capped = await readWithinLimit(response.body, maxBytes);
196
215
  if ('over' in capped) throw bodyTooLarge(url, capped.over, maxBytes);
197
216
  const body = new TextDecoder().decode(capped.bytes);
198
- return responseOver(url, response.status, headerRecord(response.headers), () =>
199
- Promise.resolve(body),
217
+ return responseOver(
218
+ url,
219
+ response.status,
220
+ headerRecord(response.headers),
221
+ () => Promise.resolve(body),
222
+ init.secrets,
200
223
  );
201
224
  } catch (thrown) {
202
225
  // A deadline that fired is this package's own timeout, with its own code and fix — never
package/src/index.ts CHANGED
@@ -18,6 +18,7 @@ export type {
18
18
  ScrapeAuth,
19
19
  } from './auth';
20
20
  export { burnSession, createPrompt, ensureAuthenticated, restorableSession } from './auth';
21
+ export { browserRecord } from './browser-record';
21
22
  export type { CaptureClip, CaptureFraming } from './capture-clip';
22
23
  export { assertCaptureFraming } from './capture-clip';
23
24
  export type {
@@ -28,7 +29,7 @@ export type {
28
29
  CdpRequestLike,
29
30
  CdpScreenshotOptions,
30
31
  } from './cdp-port';
31
- export { parseSnapshots, snapshotExpression } from './cdp-snapshot';
32
+ export { clearExpression, parseSnapshots, snapshotExpression } from './cdp-snapshot';
32
33
  export type { CdpTargetInit } from './cdp-target';
33
34
  export { CDP_DRIVER, cdpTarget } from './cdp-target';
34
35
  export type { Deadline, ScrapeClock, TestScrapeClock } from './clock';
@@ -172,9 +173,13 @@ export type { ScrapeSecrets, SecretResolver } from './secrets';
172
173
  export {
173
174
  blankPasswordFields,
174
175
  createSecretBag,
176
+ MIN_REDACTABLE_LENGTH,
175
177
  redactSecrets,
176
178
  SECRET_PLACEHOLDER,
179
+ safeConsole,
177
180
  safeHtml,
181
+ safeNetwork,
182
+ safePageErrors,
178
183
  } from './secrets';
179
184
  export type { ScrapeSessionStore, SessionSnapshot, SessionState } from './session-state';
180
185
  export {
@@ -60,6 +60,9 @@ export async function openOfflineSession(init: OfflineSessionInit): Promise<Scra
60
60
  source: init.source,
61
61
  // The same gate the page above holds, from the same field: two legs, one robots decision.
62
62
  robots: init.session.robots,
63
+ // The same bag, for the same reason: a redaction only the live leg performs is one no
64
+ // fixture can prove.
65
+ secrets: init.session.secrets,
63
66
  maxAgeMs: init.maxAgeMs,
64
67
  }),
65
68
  close: () => target.close(),
@@ -22,7 +22,7 @@ import type {
22
22
  } from './page';
23
23
  import type { RobotsGate } from './robots';
24
24
  import type { ScrapeSecrets } from './secrets';
25
- import { safeHtml } from './secrets';
25
+ import { safeConsole, safeHtml, safeNetwork, safePageErrors } from './secrets';
26
26
  import type {
27
27
  CaptureOptions,
28
28
  ElementSnapshot,
@@ -130,7 +130,16 @@ function frameOver(
130
130
  await wait(selector, options, 'actionable');
131
131
  await (await resolve()).select(selector, values);
132
132
  },
133
+ // Resolved through `resolve()` like every other verb, which is what makes a FRAME's `query`
134
+ // read the frame's document: `resolve` is `resolveChild` there, and both drivers override
135
+ // `query` on the frame target. A forward that reached for a captured page target instead
136
+ // would be the same defect `clear` had.
137
+ async query(selector): Promise<readonly ElementSnapshot[]> {
138
+ return await (await resolve()).query(selector);
139
+ },
133
140
  async values(selector): Promise<readonly ElementValue[]> {
141
+ // `resolve()` and not `this.query(…)`: a caller who destructures — `const { values } = page`
142
+ // — loses `this`, and a projection of the read above must not depend on how it was reached.
134
143
  return (await (await resolve()).query(selector)).map(toValue);
135
144
  },
136
145
  async text(selector): Promise<string> {
@@ -229,11 +238,21 @@ export function pageOverTarget(target: ScrapeTarget, ctx: PageContext): ScrapePa
229
238
  async download(options?: DownloadRequest): Promise<ScrapeDownloadFile> {
230
239
  return await target.download({ timeoutMs: options?.timeout ?? ctx.defaultTimeoutMs });
231
240
  },
241
+ // `async`, for `download()`'s reason: `ScrapeTarget` is the seam a third party implements, and
242
+ // a driver that throws synchronously from a promise-typed method would escape `.catch()`.
243
+ async offline(enabled: boolean): Promise<void> {
244
+ await target.setOfflineMode(enabled);
245
+ },
232
246
  cookies: (): Promise<readonly ScrapeCookie[]> => target.cookies(),
233
247
  session: () => target.session(),
234
- console: () => target.console.entries(),
235
- pageErrors: () => target.pageErrors.entries(),
236
- network: () => target.network.entries(),
248
+ // Redacted BY VALUE on the way out, the same pass `html()` makes. A console line and a request
249
+ // URL are two of the four surfaces `secrets.ts` names, and neither had a redaction caller: a
250
+ // login endpoint fetched with the password in its query string, or a site that logs the
251
+ // credential it rejected, put the value in `page.network()` and `page.console()` verbatim —
252
+ // and from there into `saveFailureArtifact`'s stored report.
253
+ console: () => safeConsole(target.console.entries(), ctx.secrets),
254
+ pageErrors: () => safePageErrors(target.pageErrors.entries(), ctx.secrets),
255
+ network: () => safeNetwork(target.network.entries(), ctx.secrets),
237
256
  networkDropped: () => target.network.dropped,
238
257
  pageErrorsDropped: () => target.pageErrors.dropped,
239
258
  };
package/src/page.ts CHANGED
@@ -1,6 +1,7 @@
1
- // The vocabulary a `run()` body writes against — small, declarative, and driver-blind. Fourteen
2
- // verbs, chosen because every scraper in the audit re-implemented these and nothing else; a
3
- // fifteenth would be a second way to do something on this list.
1
+ // The vocabulary a `run()` body writes against — small, declarative, and driver-blind. Every verb
2
+ // is here because every scraper in the audit re-implemented it and nothing else; a new one is a
3
+ // second way to do something already on this list. NO COUNT: this said "Fourteen verbs" against
4
+ // twenty-three, and a prose ordinal is wrong the moment the next verb lands.
4
5
  //
5
6
  // Nothing here mentions puppeteer, CDP, a frame handle or a locator. That is the seam: a run body
6
7
  // written against this file runs unchanged on a real browser, on a recorded fixture and on a
@@ -47,6 +48,17 @@ export interface ScrapeFrame {
47
48
  /** Clears first, then types — the spelling a login form wants. */
48
49
  fill(selector: string, text: string | Secret, options?: WaitOptions): Promise<void>;
49
50
  select(selector: string, values: readonly string[], options?: WaitOptions): Promise<void>;
51
+ /**
52
+ * Every match, as SNAPSHOTS — `visible`, `enabled` and (on a driver with a layout engine) the
53
+ * box and hit-target, which `values()` projects away.
54
+ *
55
+ * It exists because `ScrapeTarget.query` already answered all of that and nothing above it
56
+ * exposed it, so the first caller that needed "is this visible?" wrote its own
57
+ * `display !== 'none' && visibility !== 'hidden' && opacity !== '0'` — a second definition of
58
+ * "visible" in one framework. `values()` remains the projection for row assembly; this is the
59
+ * read for a decision about an element.
60
+ */
61
+ query(selector: string): Promise<readonly ElementSnapshot[]>;
50
62
  /** Every match, as values. Row assembly is the app's business, never the framework's. */
51
63
  values(selector: string): Promise<readonly ElementValue[]>;
52
64
  /** The first match's text, or `''`. */
@@ -90,6 +102,13 @@ export interface ScrapePage extends ScrapeFrame {
90
102
  /** The file the last click produced, or `X_SCRAPE_DOWNLOAD_TIMEOUT`. */
91
103
  download(options?: DownloadRequest): Promise<ScrapeDownloadFile>;
92
104
  cookies(): Promise<readonly ScrapeCookie[]>;
105
+ /**
106
+ * Cut the BROWSER's network, or restore it — what a PWA's offline behaviour has to be tested
107
+ * against. Refused with `X_NOT_IMPLEMENTED` on a driver that has no browser: patching `fetch`
108
+ * in the test process cannot reach a browser's own requests, so a driver that quietly answered
109
+ * "done" would let an offline assertion pass against an app that never went offline.
110
+ */
111
+ offline(enabled: boolean): Promise<void>;
93
112
  /**
94
113
  * The handoff, made explicit: what the HTTP leg will send, as a value an author can inspect and
95
114
  * a fixture can assert on. `http` uses it automatically — this is for seeing what carried over.
package/src/recording.ts CHANGED
@@ -22,11 +22,27 @@ export interface PageRecording {
22
22
  readonly recordedAt?: string | undefined;
23
23
  }
24
24
 
25
+ /**
26
+ * `.min(0)` where empty is a real recorded answer, `t.string` where it could not be — this
27
+ * package's own idiom, spelled with its reason at `cdp-target.ts` and `cdp-snapshot.ts`.
28
+ *
29
+ * `html` and a `frames` entry are the empty ones: a blank document and an empty iframe are both
30
+ * what real sites serve, and refusing them made `fixtureBrowser()` throw `X_VALIDATION_FAILED` on
31
+ * a recording `fakeBrowser()` replayed happily — the two offline drivers disagreeing about the
32
+ * same file, which is the one thing a shared format exists to prevent.
33
+ *
34
+ * `url`, an `evaluate` entry and a `downloads` entry stay non-empty, each for a reason: there is
35
+ * nothing to key a recording by without a URL, an `evaluate` value is JSON TEXT and `''` is not
36
+ * JSON (`htmlTarget` would throw a bare `SyntaxError` at read time instead), and a download is
37
+ * `filename:contents`.
38
+ */
39
+ const recordedText = t.string.min(0);
40
+
25
41
  export const pageRecordingSchema: StandardSchemaV1<unknown, PageRecording> = t.object({
26
42
  url: t.string,
27
- html: t.string,
43
+ html: recordedText,
28
44
  evaluate: t.optional(t.record(t.string)),
29
- frames: t.optional(t.record(t.string)),
45
+ frames: t.optional(t.record(recordedText)),
30
46
  downloads: t.optional(t.record(t.string)),
31
47
  recordedAt: t.optional(t.string),
32
48
  }) as unknown as StandardSchemaV1<unknown, PageRecording>;
@@ -49,12 +65,17 @@ export interface HttpRecording {
49
65
  readonly recordedAt?: string | undefined;
50
66
  }
51
67
 
68
+ /**
69
+ * `body` and a header value are the empty ones. A 204 has no body BY DEFINITION — refusing it
70
+ * meant the one status a `POST /like` answers could not be recorded at all — and a site sending
71
+ * `x-trace:` with nothing after it is a header it sent.
72
+ */
52
73
  export const httpRecordingSchema: StandardSchemaV1<unknown, HttpRecording> = t.object({
53
74
  url: t.string,
54
75
  method: t.string,
55
76
  status: t.number,
56
- body: t.string,
57
- headers: t.optional(t.record(t.string)),
77
+ body: recordedText,
78
+ headers: t.optional(t.record(recordedText)),
58
79
  recordedAt: t.optional(t.string),
59
80
  }) as unknown as StandardSchemaV1<unknown, HttpRecording>;
60
81
 
package/src/rings.ts CHANGED
@@ -2,6 +2,8 @@
2
2
  // thousand pages that kept every console line and every request holds the run's entire browsing
3
3
  // history in the worker's heap, and the incident is an OOM two hours in rather than a scraper bug.
4
4
 
5
+ import { assert } from '@ultimat3/core';
6
+
5
7
  export interface ConsoleLine {
6
8
  readonly level: 'log' | 'info' | 'warn' | 'error' | 'debug';
7
9
  readonly text: string;
@@ -99,7 +101,25 @@ export type ConsoleRing = Ring<ConsoleLine>;
99
101
  export type NetworkRing = Ring<NetworkEntry>;
100
102
  export type PageErrorRing = Ring<PageError>;
101
103
 
104
+ /**
105
+ * REFUSED rather than clamped, the shape `@ultimat3/storage`'s `resolveListLimit` uses — a bound
106
+ * with no code of its own is still a coded refusal, never a bare `Error`.
107
+ *
108
+ * Both wrong values fail in opposite directions and neither is survivable. A NEGATIVE capacity
109
+ * makes `while (items.length > capacity) items.shift()` spin forever on an already-empty array
110
+ * (`0 > -1`): `createRing(-1).push(1)` never returns, and it is a synchronous loop on the worker's
111
+ * only thread, inside a `page.on('request')` handler — past `ctx.signal`, past the wedge watchdog
112
+ * and past the job timeout, which is incident #1 in this file's header. `NaN` fails the other way:
113
+ * the comparison is false forever, the ring is UNBOUNDED, and the OOM the bound exists to prevent
114
+ * arrives two hours in. Zero is a ring that discards everything silently, and a fraction is a
115
+ * bound that is not the number anybody asked for.
116
+ */
102
117
  export function createRing<T>(capacity: number = DEFAULT_RING_CAPACITY): Ring<T> {
118
+ assert(
119
+ Number.isSafeInteger(capacity) && capacity > 0,
120
+ `a ring capacity must be a positive integer, got ${String(capacity)}: a negative one spins forever on push() and NaN makes the ring unbounded`,
121
+ `pass a positive capacity — createRing(${String(DEFAULT_RING_CAPACITY)}) — or omit it and take DEFAULT_RING_CAPACITY`,
122
+ );
103
123
  const items: T[] = [];
104
124
  let dropped = 0;
105
125
  return {
package/src/secrets.ts CHANGED
@@ -8,6 +8,7 @@
8
8
 
9
9
  import type { Secret } from '@ultimat3/core';
10
10
  import { revealSecret, secret, UltimateError } from '@ultimat3/core';
11
+ import type { ConsoleLine, NetworkEntry, PageError } from './rings';
11
12
 
12
13
  export interface ScrapeSecrets {
13
14
  readonly names: readonly string[];
@@ -61,9 +62,33 @@ export function createSecretBag(
61
62
  export const SECRET_PLACEHOLDER = '[redacted]';
62
63
 
63
64
  /**
64
- * Redaction BY VALUE, over any text this package is about to persist: page HTML, a console line,
65
- * a request URL, an error cause. Name-based redaction only catches a secret travelling under a
66
- * name somebody remembered to list, and a password pasted into a query string travels under none.
65
+ * The floor, and it is a real hole rather than an implementation detail: a secret shorter than
66
+ * this is NOT redacted.
67
+ *
68
+ * A three-character value is a substring of ordinary prose — a PIN of `123` would blank every
69
+ * price, every id and every timestamp fragment on the page, and an artifact redacted into
70
+ * unreadability is one nobody can diagnose from. So the bound is deliberate, and the answer for a
71
+ * genuinely short credential is that it must not be one: `page.type(selector, secrets.get(name))`
72
+ * still TAINTS the page whatever the length, so `screenshot()` and `pdf()` are refused either way.
73
+ * Only the by-value text pass below skips it. Exported so a caller can check rather than discover.
74
+ */
75
+ export const MIN_REDACTABLE_LENGTH = 4;
76
+
77
+ /**
78
+ * Redaction BY VALUE, over the text this package hands back or persists. Name-based redaction only
79
+ * catches a secret travelling under a name somebody remembered to list, and a password pasted into
80
+ * a query string travels under none.
81
+ *
82
+ * The four surfaces, and they are the four callers below — page HTML (`safeHtml`), a console line
83
+ * (`safeConsole`), a request URL (`safeNetwork`) and a thrown message (`safePageErrors`), plus the
84
+ * HTTP leg's own `X_SCRAPE_HTTP_FAILED` cause, which redacts at its throw site in `http.ts`
85
+ * because the body is gone by the time anything else could. Three of those had NO caller until
86
+ * 2026-08-24 while this header claimed all four.
87
+ *
88
+ * What is still NOT redacted, stated rather than implied: a URL in any other error's `cause` or
89
+ * `meta` (`X_SCRAPE_HOST_BLOCKED`, `X_SCRAPE_FIXTURE_MISSING`), a value shorter than
90
+ * `MIN_REDACTABLE_LENGTH`, and pixels — which is why a secret TAINTS the page and captures are
91
+ * refused outright.
67
92
  *
68
93
  * Longest first, so a secret that contains another one does not leave its tail behind.
69
94
  */
@@ -71,13 +96,53 @@ export function redactSecrets(text: string, secrets: ScrapeSecrets | undefined):
71
96
  if (secrets === undefined || secrets.names.length === 0) return text;
72
97
  const values = secrets.names
73
98
  .map((name) => revealSecret(secrets.get(name)))
74
- .filter((value) => value.length >= 4)
99
+ .filter((value) => value.length >= MIN_REDACTABLE_LENGTH)
75
100
  .sort((a, b) => b.length - a.length);
76
101
  let out = text;
77
102
  for (const value of values) out = out.split(value).join(SECRET_PLACEHOLDER);
78
103
  return out;
79
104
  }
80
105
 
106
+ /**
107
+ * The bounded tails, redacted on the way OUT of the page vocabulary rather than on the way into
108
+ * the ring. A driver fills its rings from the browser's own events, and a `type()` that has not
109
+ * happened yet cannot redact a line already recorded — so the pass belongs where the entries are
110
+ * read, which is also the one place every driver shares.
111
+ *
112
+ * Each returns the entry UNCHANGED when nothing matched, so a run with no secrets declared pays
113
+ * one comparison per entry and allocates nothing.
114
+ */
115
+ export const safeConsole = (
116
+ lines: readonly ConsoleLine[],
117
+ secrets: ScrapeSecrets | undefined,
118
+ ): readonly ConsoleLine[] =>
119
+ lines.map((line) => {
120
+ const text = redactSecrets(line.text, secrets);
121
+ return text === line.text ? line : { ...line, text };
122
+ });
123
+
124
+ /** A password pasted into a query string is the case name-based redaction cannot see at all. */
125
+ export const safeNetwork = (
126
+ entries: readonly NetworkEntry[],
127
+ secrets: ScrapeSecrets | undefined,
128
+ ): readonly NetworkEntry[] =>
129
+ entries.map((entry) => {
130
+ const url = redactSecrets(entry.url, secrets);
131
+ return url === entry.url ? entry : { ...entry, url };
132
+ });
133
+
134
+ /** The stack too: a framework that prints the argument it threw on puts the value in both. */
135
+ export const safePageErrors = (
136
+ errors: readonly PageError[],
137
+ secrets: ScrapeSecrets | undefined,
138
+ ): readonly PageError[] =>
139
+ errors.map((error) => {
140
+ const message = redactSecrets(error.message, secrets);
141
+ const stack = error.stack === undefined ? undefined : redactSecrets(error.stack, secrets);
142
+ if (message === error.message && stack === error.stack) return error;
143
+ return { ...error, message, ...(stack === undefined ? {} : { stack }) };
144
+ });
145
+
81
146
  /** Every `<input …>` tag, whole, so the rewrite below never has to reason about attribute order. */
82
147
  const INPUT_TAG = /<input\b[^>]*>/gi;
83
148
  /** `type=password`, quoted either way or bare. */