@ultimat3/scraping 5.0.1 → 7.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -84,6 +84,29 @@ Both legs replay from **one** fixture directory (`fixtureBrowser(dir)`), so a hy
84
84
  login, session handoff, HTTP bulk fetch — is tested end to end. Both legs apply the same robots
85
85
  gate, offline included.
86
86
 
87
+ ## What the page reports back, and the one thing a picture cannot say
88
+
89
+ Three bounded rings, read off `ScrapePage`. Bounded because a ten-thousand-page run that kept
90
+ every line holds the whole browsing history in the worker's heap — and each read says how much the
91
+ bound threw away, so a count taken from one is never quietly a floor.
92
+
93
+ | Read | Answers |
94
+ |---|---|
95
+ | `page.console()` | `ConsoleLine[]` — what the page LOGGED |
96
+ | `page.pageErrors()` / `page.pageErrorsDropped()` | `PageError[]` — what the page THREW and nobody caught, with the `stack` when the exception carried one |
97
+ | `page.network()` / `page.networkDropped()` | `NetworkEntry[]` — every request, refusals included |
98
+
99
+ `console()` and `pageErrors()` are separate streams because they are separate events: an island
100
+ that throws during hydration calls no console method, so a scrape reading the console alone sees a
101
+ page that looks silent and is broken — and a screenshot of it is a picture of the server-rendered
102
+ markup, indistinguishable from one that worked.
103
+
104
+ Empty is a legitimate answer, never a missing method: the offline drivers parse markup and execute
105
+ none of it, so nothing there can throw. `ScrapeTarget.pageErrors` is a **required** ring for the
106
+ same reason — a third-party driver that could omit it would be silent about errors it can see.
107
+ Entries are built with `pageErrorEntry()`, which truncates at `MAX_PAGE_ERROR_CHARS`: the ring
108
+ bounds the count, and one `Maximum call stack size exceeded` is thousands of frames.
109
+
87
110
  ## What it owns
88
111
 
89
112
  | Module | Owns |
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ultimat3/scraping",
3
- "version": "5.0.1",
3
+ "version": "7.0.0",
4
4
  "description": "Browser automation as a job: scrape() returns a JobHandle",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -30,9 +30,9 @@
30
30
  "test": "bun test"
31
31
  },
32
32
  "dependencies": {
33
- "@ultimat3/core": "5.0.1",
34
- "@ultimat3/jobs": "5.0.1",
35
- "@ultimat3/schema": "5.0.1",
36
- "@ultimat3/storage": "5.0.1"
33
+ "@ultimat3/core": "7.0.0",
34
+ "@ultimat3/jobs": "7.0.0",
35
+ "@ultimat3/schema": "7.0.0",
36
+ "@ultimat3/storage": "7.0.0"
37
37
  }
38
38
  }
package/src/cdp-fake.ts CHANGED
@@ -36,6 +36,12 @@ type Handlers = Map<string, ((payload: unknown) => void)[]>;
36
36
  export interface FakeCdpBrowser extends CdpBrowserLike {
37
37
  /** Fire a request event, as a real browser would when the page fetches a subresource. */
38
38
  emitRequest(url: string, resourceType: string): void;
39
+ /**
40
+ * Fire `pageerror` — the page threw and nothing caught it. `payload` is whatever the library
41
+ * would hand a handler, `unknown` on purpose: a page can throw a string as easily as an `Error`,
42
+ * and the target reads it defensively either way.
43
+ */
44
+ emitPageError(payload: unknown): void;
39
45
  readonly aborted: readonly string[];
40
46
  readonly closed: boolean;
41
47
  }
@@ -130,6 +136,9 @@ export function fakeCdpBrowser(init: FakeCdpPageInit): FakeCdpBrowser {
130
136
  });
131
137
  }
132
138
  },
139
+ emitPageError(payload: unknown): void {
140
+ for (const handler of handlers.get('pageerror') ?? []) handler(payload);
141
+ },
133
142
  aborted,
134
143
  get closed(): boolean {
135
144
  return closed;
package/src/cdp-port.ts CHANGED
@@ -37,6 +37,17 @@ export interface CdpPageLike {
37
37
  screenshot(options: { readonly fullPage?: boolean }): Promise<Uint8Array | string>;
38
38
  pdf(options?: Record<string, unknown>): Promise<Uint8Array>;
39
39
  setRequestInterception(enabled: boolean): Promise<void>;
40
+ /**
41
+ * `event` stays a bare `string` — a union of the four names this package subscribes to would be
42
+ * this file naming somebody else's event vocabulary, which is the thing it exists not to do, and
43
+ * a launcher whose emitter is wider would then fail to satisfy the port for no reason.
44
+ *
45
+ * Those four, and the pair that is easy to confuse: `request`, `console`, `pageerror` — an
46
+ * uncaught exception INSIDE the page, which leaves the session perfectly usable — and `error`,
47
+ * which is the renderer CRASHING and is what `X_SCRAPE_PAGE_CRASHED` is raised from. Every
48
+ * payload arrives `unknown` and is read defensively in `cdp-target.ts`; nothing here may name
49
+ * the library's own types for them.
50
+ */
40
51
  on(event: string, handler: (payload: unknown) => void): unknown;
41
52
  frames(): readonly CdpFrameLike[];
42
53
  close(): Promise<void>;
package/src/cdp-target.ts CHANGED
@@ -10,8 +10,16 @@ import type { ScrapeClock } from './clock';
10
10
  import { browserUnreachable, pageCrashed, scrapeNotImplemented } from './error-throws';
11
11
  import type { InterceptRules } from './intercept';
12
12
  import { interceptVerdict, refusalEntry } from './intercept';
13
- import type { ConsoleLine, NetworkEntry, ResourceType } from './rings';
14
- import { createRing, RESOURCE_TYPES } from './rings';
13
+ import type {
14
+ ConsoleLine,
15
+ ConsoleRing,
16
+ NetworkEntry,
17
+ NetworkRing,
18
+ PageError,
19
+ PageErrorRing,
20
+ ResourceType,
21
+ } from './rings';
22
+ import { createRing, pageErrorEntry, RESOURCE_TYPES } from './rings';
15
23
  import type { SessionSnapshot } from './session-state';
16
24
  import type {
17
25
  CaptureOptions,
@@ -88,6 +96,25 @@ const readStringFrom = (owner: unknown, key: string): string | undefined => {
88
96
  return typeof answer === 'string' ? answer : undefined;
89
97
  };
90
98
 
99
+ /**
100
+ * A `pageerror` payload, read defensively — never cast, and never assumed to be an `Error`.
101
+ *
102
+ * `readStringFrom`, the same reader the console handler uses, because the payload has the same
103
+ * problem: `message` and `stack` are an own property on one build and an accessor on another, and
104
+ * a schema parse cannot call an accessor. A page can also `throw 'a string'` or throw a frozen
105
+ * object with no `message` at all — both reach here, and both are recorded as SOMETHING having
106
+ * thrown, because an entry with a poor message is still the difference between "the island threw"
107
+ * and silence.
108
+ */
109
+ const readPageError = (payload: unknown, at: number): PageError => {
110
+ if (typeof payload === 'string') return pageErrorEntry({ message: payload, at });
111
+ return pageErrorEntry({
112
+ message: readStringFrom(payload, 'message') ?? '',
113
+ stack: readStringFrom(payload, 'stack'),
114
+ at,
115
+ });
116
+ };
117
+
91
118
  /**
92
119
  * The one failure `guard()` must NOT re-label, and the line is drawn at exactly one code.
93
120
  *
@@ -116,17 +143,25 @@ export interface CdpTargetInit {
116
143
  readonly ringCapacity?: number | undefined;
117
144
  }
118
145
 
146
+ /**
147
+ * Everything `arm()` writes into. Named rather than positional: three rings of near-identical
148
+ * type plus a latch is a call site nobody can read, and swapping two of them is a mistake the
149
+ * compiler cannot catch.
150
+ */
151
+ interface CdpSinks {
152
+ readonly network: NetworkRing;
153
+ readonly console: ConsoleRing;
154
+ readonly pageErrors: PageErrorRing;
155
+ readonly crashed: { value: string | undefined };
156
+ }
157
+
119
158
  /**
120
159
  * Interception is armed BEFORE the first navigation and refuses at the request, not after the
121
160
  * response — an `allowHosts` that reported afterwards would be a log line about bytes that
122
161
  * already left the container.
123
162
  */
124
- async function arm(
125
- init: CdpTargetInit,
126
- network: ReturnType<typeof createRing<NetworkEntry>>,
127
- console_: ReturnType<typeof createRing<ConsoleLine>>,
128
- crashed: { value: string | undefined },
129
- ): Promise<void> {
163
+ async function arm(init: CdpTargetInit, sinks: CdpSinks): Promise<void> {
164
+ const { network, console: console_, pageErrors, crashed } = sinks;
130
165
  await init.page.setRequestInterception(true);
131
166
  init.page.on('request', (payload) => {
132
167
  const request = asRequest(payload);
@@ -154,6 +189,23 @@ async function arm(
154
189
  at: init.clock.now().getTime(),
155
190
  });
156
191
  });
192
+ /**
193
+ * The page threw and nothing caught it. THE gap this ring closes: a screenshot of an island
194
+ * that threw during hydration is a picture of the server-rendered markup, indistinguishable
195
+ * from a page that worked — and `console` does not carry it, because throwing calls no console
196
+ * method. Subscribed here, beside the others, so a target is observing before its first
197
+ * navigation: an exception raised during load has no second chance to be recorded.
198
+ *
199
+ * NOT the same event as `error` below, and the difference is the whole reason this is a
200
+ * separate handler: puppeteer's `pageerror` is "an uncaught exception happens within the page"
201
+ * and its `error` is "the page crashes" (`PageEvent.PageError` / `PageEvent.Error`). One is the
202
+ * app being broken and the session is fine; the other is the tab being gone. Recording a
203
+ * `pageerror` into `crashed` would make every scrape of a page with one bad island answer
204
+ * X_SCRAPE_PAGE_CRASHED — a code registered `terminal` — for a page still perfectly usable.
205
+ */
206
+ init.page.on('pageerror', (payload) => {
207
+ pageErrors.push(readPageError(payload, init.clock.now().getTime()));
208
+ });
157
209
  // A renderer that dies must be a CODE, not a hang: every later call answers X_SCRAPE_PAGE_CRASHED
158
210
  // instead of waiting out its own timeout against a tab that is gone.
159
211
  init.page.on('error', (payload) => {
@@ -164,8 +216,9 @@ async function arm(
164
216
  export async function cdpTarget(init: CdpTargetInit): Promise<ScrapeTarget> {
165
217
  const console_ = createRing<ConsoleLine>(init.ringCapacity);
166
218
  const network = createRing<NetworkEntry>(init.ringCapacity);
219
+ const pageErrors = createRing<PageError>(init.ringCapacity);
167
220
  const crashed: { value: string | undefined } = { value: undefined };
168
- await arm(init, network, console_, crashed);
221
+ await arm(init, { network, console: console_, pageErrors, crashed });
169
222
  let pendingStorage: SessionSnapshot | undefined;
170
223
 
171
224
  const originOf = (url: string): string => {
@@ -233,6 +286,9 @@ export async function cdpTarget(init: CdpTargetInit): Promise<ScrapeTarget> {
233
286
  driver: CDP_DRIVER,
234
287
  console: console_,
235
288
  network,
289
+ // Shared with every frame target below, through the spread: an exception is the PAGE's, and a
290
+ // per-frame ring would hide a throw from an iframe behind whichever handle the caller held.
291
+ pageErrors,
236
292
  url: () => init.page.url(),
237
293
  goto: (url: string, options: GotoOptions) =>
238
294
  guard('goto', async () => {
@@ -13,7 +13,7 @@ import type { InterceptRules } from './intercept';
13
13
  import { interceptVerdict, refusalEntry } from './intercept';
14
14
  import type { PageRecording } from './recording';
15
15
  import { splitDownload } from './recording';
16
- import type { ConsoleLine, NetworkEntry } from './rings';
16
+ import type { ConsoleLine, NetworkEntry, PageError } from './rings';
17
17
  import { createRing } from './rings';
18
18
  import type { SessionSnapshot } from './session-state';
19
19
  import { EMPTY_SESSION } from './session-state';
@@ -74,6 +74,12 @@ const keyOf = (selector: string, element: ElementSnapshot | undefined): string =
74
74
  export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
75
75
  const consoleRing = createRing<ConsoleLine>();
76
76
  const networkRing = createRing<NetworkEntry>();
77
+ // Built and never pushed to, deliberately: this target parses markup and executes none of it, so
78
+ // there is no uncaught exception for it to have. It ANSWERS rather than omitting the ring —
79
+ // `page.pageErrors()` returning `[]` here is the honest "nothing threw, and nothing could",
80
+ // where a missing ring would be a page method that throws on two of the three drivers.
81
+ // `driver-parity.test.ts` pins the divergence, beside the box/hit-target one it already carries.
82
+ const pageErrorRing = createRing<PageError>();
77
83
  const overlay = new Map<string, string>();
78
84
  let page: PageRecording = init.start ?? EMPTY;
79
85
  let armed: string | undefined;
@@ -164,6 +170,7 @@ export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
164
170
  driver: init.driver,
165
171
  console: consoleRing,
166
172
  network: networkRing,
173
+ pageErrors: pageErrorRing,
167
174
  url: () => page.url,
168
175
  goto: (url: string, _options: GotoOptions): Promise<void> => navigate(url),
169
176
  content: (): Promise<string> => {
package/src/index.ts CHANGED
@@ -137,10 +137,18 @@ export type {
137
137
  ConsoleRing,
138
138
  NetworkEntry,
139
139
  NetworkRing,
140
+ PageError,
141
+ PageErrorRing,
140
142
  ResourceType,
141
143
  Ring,
142
144
  } from './rings';
143
- export { createRing, DEFAULT_RING_CAPACITY, RESOURCE_TYPES } from './rings';
145
+ export {
146
+ createRing,
147
+ DEFAULT_RING_CAPACITY,
148
+ MAX_PAGE_ERROR_CHARS,
149
+ pageErrorEntry,
150
+ RESOURCE_TYPES,
151
+ } from './rings';
144
152
  export type { RobotsFetch, RobotsGate, RobotsGateInit, RobotsPolicy, RobotsRules } from './robots';
145
153
  export { createRobotsGate, DEFAULT_ROBOTS_AGENT, parseRobots, robotsAllows } from './robots';
146
154
  export type { RobotsFetchInit } from './robots-fetch';
@@ -212,7 +212,9 @@ export function pageOverTarget(target: ScrapeTarget, ctx: PageContext): ScrapePa
212
212
  cookies: (): Promise<readonly ScrapeCookie[]> => target.cookies(),
213
213
  session: () => target.session(),
214
214
  console: () => target.console.entries(),
215
+ pageErrors: () => target.pageErrors.entries(),
215
216
  network: () => target.network.entries(),
216
217
  networkDropped: () => target.network.dropped,
218
+ pageErrorsDropped: () => target.pageErrors.dropped,
217
219
  };
218
220
  }
package/src/page.ts CHANGED
@@ -8,7 +8,7 @@
8
8
 
9
9
  import type { Secret } from '@ultimat3/core';
10
10
  import type { ActionabilityState } from './actionability';
11
- import type { ConsoleLine, NetworkEntry } from './rings';
11
+ import type { ConsoleLine, NetworkEntry, PageError } from './rings';
12
12
  import type { SessionSnapshot } from './session-state';
13
13
  import type { ElementSnapshot, ScrapeCookie, ScrapeDownloadFile } from './target';
14
14
 
@@ -93,6 +93,13 @@ export interface ScrapePage extends ScrapeFrame {
93
93
  session(): Promise<SessionSnapshot>;
94
94
  /** The bounded tail. Bounded because a long run's full history is an OOM, not a log. */
95
95
  console(): readonly ConsoleLine[];
96
+ /**
97
+ * The uncaught exceptions the page threw, which `console()` does NOT carry: an island that
98
+ * throws calls no console method, so a scrape reading console alone sees a page that looks
99
+ * silent and is broken. Empty on a driver with no JS engine — the offline drivers parse HTML
100
+ * and never execute it, so nothing there can throw.
101
+ */
102
+ pageErrors(): readonly PageError[];
96
103
  network(): readonly NetworkEntry[];
97
104
  /**
98
105
  * How many entries the bound above threw away. It is the same honesty `Ring.dropped` carries:
@@ -100,4 +107,10 @@ export interface ScrapePage extends ScrapeFrame {
100
107
  * is non-zero, and a scrape that blocked 5,000 images otherwise reports 200 with no hint.
101
108
  */
102
109
  networkDropped(): number;
110
+ /**
111
+ * The same honesty for the errors, and it is the count a verdict gates on: an island throwing
112
+ * inside a `requestAnimationFrame` produces thousands, and "3 page errors" read off a bounded
113
+ * tail of 200 would be a number a reader trusts and should not.
114
+ */
115
+ pageErrorsDropped(): number;
103
116
  }
package/src/rings.ts CHANGED
@@ -1,6 +1,6 @@
1
- // Bounded console and network history. BOUNDED is the whole point: a scrape of ten thousand
2
- // pages that kept every console line and every request holds the run's entire browsing history in
3
- // the worker's heap, and the incident is an OOM two hours in rather than a scraper bug.
1
+ // Bounded console, page-error and network history. BOUNDED is the whole point: a scrape of ten
2
+ // thousand pages that kept every console line and every request holds the run's entire browsing
3
+ // history in the worker's heap, and the incident is an OOM two hours in rather than a scraper bug.
4
4
 
5
5
  export interface ConsoleLine {
6
6
  readonly level: 'log' | 'info' | 'warn' | 'error' | 'debug';
@@ -19,6 +19,55 @@ export interface NetworkEntry {
19
19
  readonly refused?: 'blocked' | 'host' | 'robots' | undefined;
20
20
  }
21
21
 
22
+ /**
23
+ * An uncaught exception the PAGE threw — the half a picture cannot carry.
24
+ *
25
+ * Its own entry type, and not a `ConsoleLine` with `level: 'error'`, because it is a different
26
+ * event: a page that threw called no console method, so folding one into the console stream
27
+ * reports a `console.error` that never happened — and `ConsoleLine` has nowhere to keep the stack,
28
+ * which is the field that says WHICH island threw.
29
+ */
30
+ export interface PageError {
31
+ /** The exception's message. `''` when the payload carried none — never a fabricated one. */
32
+ readonly message: string;
33
+ /**
34
+ * The JS stack, when the exception carried one. Absent means the payload had none, never that
35
+ * the throw had no origin: a `throw 'a string'` in the page reaches here with a message alone.
36
+ */
37
+ readonly stack?: string | undefined;
38
+ readonly at: number;
39
+ }
40
+
41
+ /**
42
+ * The per-entry cap, which the ring itself cannot give: a ring bounds the COUNT, and one
43
+ * `RangeError: Maximum call stack size exceeded` carries thousands of frames — 200 of those is
44
+ * megabytes held per page, which is the same OOM this file exists to prevent, one level down.
45
+ */
46
+ export const MAX_PAGE_ERROR_CHARS = 4_000;
47
+
48
+ const clamped = (text: string): string =>
49
+ text.length <= MAX_PAGE_ERROR_CHARS ? text : `${text.slice(0, MAX_PAGE_ERROR_CHARS - 1)}…`;
50
+
51
+ /**
52
+ * The ONE way a `PageError` is built. A driver that assembled the object literal itself would be
53
+ * a driver whose stacks are unbounded, and the truncation would then be a rule per driver rather
54
+ * than a property of the type.
55
+ */
56
+ export function pageErrorEntry(input: {
57
+ readonly message: string;
58
+ readonly stack?: string | undefined;
59
+ readonly at: number;
60
+ }): PageError {
61
+ // An empty stack is ABSENT, not empty: `stack: ''` prints as a blank stack, which reads as
62
+ // "the exception had no origin" rather than "the payload did not carry one".
63
+ const stack = input.stack === undefined || input.stack === '' ? undefined : clamped(input.stack);
64
+ return {
65
+ message: clamped(input.message),
66
+ ...(stack === undefined ? {} : { stack }),
67
+ at: input.at,
68
+ };
69
+ }
70
+
22
71
  export const RESOURCE_TYPES = [
23
72
  'document',
24
73
  'stylesheet',
@@ -48,6 +97,7 @@ export interface Ring<T> {
48
97
 
49
98
  export type ConsoleRing = Ring<ConsoleLine>;
50
99
  export type NetworkRing = Ring<NetworkEntry>;
100
+ export type PageErrorRing = Ring<PageError>;
51
101
 
52
102
  export function createRing<T>(capacity: number = DEFAULT_RING_CAPACITY): Ring<T> {
53
103
  const items: T[] = [];
package/src/target.ts CHANGED
@@ -3,7 +3,7 @@
3
3
  // lets `page-over-target.ts` be the ONE implementation of `ScrapePage` for the real browser, the
4
4
  // fixture replayer and the fake alike.
5
5
 
6
- import type { ConsoleRing, NetworkRing } from './rings';
6
+ import type { ConsoleRing, NetworkRing, PageErrorRing } from './rings';
7
7
  import type { SessionSnapshot } from './session-state';
8
8
 
9
9
  /**
@@ -96,6 +96,15 @@ export interface ScrapeTarget {
96
96
  readonly driver: string;
97
97
  readonly console: ConsoleRing;
98
98
  readonly network: NetworkRing;
99
+ /**
100
+ * REQUIRED, and empty is a legitimate answer. A driver with no JS engine has no uncaught
101
+ * exception to report and answers an empty ring; an optional member would instead let a driver
102
+ * be silent about errors it CAN see, which is precisely the blind spot this ring closes — and
103
+ * `ScrapePage.pageErrors()` would then be a method that answers nothing on an unknown subset of
104
+ * drivers. A third-party driver added before 2026-08-21 gets a type error naming this field,
105
+ * which is the whole enforcement.
106
+ */
107
+ readonly pageErrors: PageErrorRing;
99
108
  url(): string;
100
109
  goto(url: string, options: GotoOptions): Promise<void>;
101
110
  /** Serialised HTML of THIS target — the document for a page, the subtree for a frame. */