@ultimat3/scraping 6.0.0 → 8.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -1,7 +1,8 @@
1
1
  # @ultimat3/scraping
2
2
 
3
- Browser automation as a **job**. `scrape()` returns a `JobHandle` — the rule's fourth instance
4
- after `llm()` (an action factory) and `backfill()` (a job factory). There is no ninth primitive.
3
+ Browser automation as a **job**. `scrape()` returns a `JobHandle` — one row of
4
+ `PRIMITIVE_FACTORIES` in `@ultimat3/core`, the derived list of every factory that ships. There is
5
+ no ninth primitive, and no ordinal here to go stale when the next factory lands.
5
6
 
6
7
  ```ts
7
8
  import { t } from '@ultimat3/schema';
@@ -84,6 +85,29 @@ Both legs replay from **one** fixture directory (`fixtureBrowser(dir)`), so a hy
84
85
  login, session handoff, HTTP bulk fetch — is tested end to end. Both legs apply the same robots
85
86
  gate, offline included.
86
87
 
88
+ ## What the page reports back, and the one thing a picture cannot say
89
+
90
+ Three bounded rings, read off `ScrapePage`. Bounded because a ten-thousand-page run that kept
91
+ every line holds the whole browsing history in the worker's heap — and each read says how much the
92
+ bound threw away, so a count taken from one is never quietly a floor.
93
+
94
+ | Read | Answers |
95
+ |---|---|
96
+ | `page.console()` | `ConsoleLine[]` — what the page LOGGED |
97
+ | `page.pageErrors()` / `page.pageErrorsDropped()` | `PageError[]` — what the page THREW and nobody caught, with the `stack` when the exception carried one |
98
+ | `page.network()` / `page.networkDropped()` | `NetworkEntry[]` — every request, refusals included |
99
+
100
+ `console()` and `pageErrors()` are separate streams because they are separate events: an island
101
+ that throws during hydration calls no console method, so a scrape reading the console alone sees a
102
+ page that looks silent and is broken — and a screenshot of it is a picture of the server-rendered
103
+ markup, indistinguishable from one that worked.
104
+
105
+ Empty is a legitimate answer, never a missing method: the offline drivers parse markup and execute
106
+ none of it, so nothing there can throw. `ScrapeTarget.pageErrors` is a **required** ring for the
107
+ same reason — a third-party driver that could omit it would be silent about errors it can see.
108
+ Entries are built with `pageErrorEntry()`, which truncates at `MAX_PAGE_ERROR_CHARS`: the ring
109
+ bounds the count, and one `Maximum call stack size exceeded` is thousands of frames.
110
+
87
111
  ## What it owns
88
112
 
89
113
  | Module | Owns |
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ultimat3/scraping",
3
- "version": "6.0.0",
3
+ "version": "8.0.0",
4
4
  "description": "Browser automation as a job: scrape() returns a JobHandle",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -30,9 +30,9 @@
30
30
  "test": "bun test"
31
31
  },
32
32
  "dependencies": {
33
- "@ultimat3/core": "6.0.0",
34
- "@ultimat3/jobs": "6.0.0",
35
- "@ultimat3/schema": "6.0.0",
36
- "@ultimat3/storage": "6.0.0"
33
+ "@ultimat3/core": "8.0.0",
34
+ "@ultimat3/jobs": "8.0.0",
35
+ "@ultimat3/schema": "8.0.0",
36
+ "@ultimat3/storage": "8.0.0"
37
37
  }
38
38
  }
package/src/cdp-fake.ts CHANGED
@@ -36,6 +36,12 @@ type Handlers = Map<string, ((payload: unknown) => void)[]>;
36
36
  export interface FakeCdpBrowser extends CdpBrowserLike {
37
37
  /** Fire a request event, as a real browser would when the page fetches a subresource. */
38
38
  emitRequest(url: string, resourceType: string): void;
39
+ /**
40
+ * Fire `pageerror` — the page threw and nothing caught it. `payload` is whatever the library
41
+ * would hand a handler, `unknown` on purpose: a page can throw a string as easily as an `Error`,
42
+ * and the target reads it defensively either way.
43
+ */
44
+ emitPageError(payload: unknown): void;
39
45
  readonly aborted: readonly string[];
40
46
  readonly closed: boolean;
41
47
  }
@@ -130,6 +136,9 @@ export function fakeCdpBrowser(init: FakeCdpPageInit): FakeCdpBrowser {
130
136
  });
131
137
  }
132
138
  },
139
+ emitPageError(payload: unknown): void {
140
+ for (const handler of handlers.get('pageerror') ?? []) handler(payload);
141
+ },
133
142
  aborted,
134
143
  get closed(): boolean {
135
144
  return closed;
package/src/cdp-port.ts CHANGED
@@ -37,6 +37,17 @@ export interface CdpPageLike {
37
37
  screenshot(options: { readonly fullPage?: boolean }): Promise<Uint8Array | string>;
38
38
  pdf(options?: Record<string, unknown>): Promise<Uint8Array>;
39
39
  setRequestInterception(enabled: boolean): Promise<void>;
40
+ /**
41
+ * `event` stays a bare `string` — a union of the four names this package subscribes to would be
42
+ * this file naming somebody else's event vocabulary, which is the thing it exists not to do, and
43
+ * a launcher whose emitter is wider would then fail to satisfy the port for no reason.
44
+ *
45
+ * Those four, and the pair that is easy to confuse: `request`, `console`, `pageerror` — an
46
+ * uncaught exception INSIDE the page, which leaves the session perfectly usable — and `error`,
47
+ * which is the renderer CRASHING and is what `X_SCRAPE_PAGE_CRASHED` is raised from. Every
48
+ * payload arrives `unknown` and is read defensively in `cdp-target.ts`; nothing here may name
49
+ * the library's own types for them.
50
+ */
40
51
  on(event: string, handler: (payload: unknown) => void): unknown;
41
52
  frames(): readonly CdpFrameLike[];
42
53
  close(): Promise<void>;
package/src/cdp-target.ts CHANGED
@@ -10,8 +10,16 @@ import type { ScrapeClock } from './clock';
10
10
  import { browserUnreachable, pageCrashed, scrapeNotImplemented } from './error-throws';
11
11
  import type { InterceptRules } from './intercept';
12
12
  import { interceptVerdict, refusalEntry } from './intercept';
13
- import type { ConsoleLine, NetworkEntry, ResourceType } from './rings';
14
- import { createRing, RESOURCE_TYPES } from './rings';
13
+ import type {
14
+ ConsoleLine,
15
+ ConsoleRing,
16
+ NetworkEntry,
17
+ NetworkRing,
18
+ PageError,
19
+ PageErrorRing,
20
+ ResourceType,
21
+ } from './rings';
22
+ import { createRing, pageErrorEntry, RESOURCE_TYPES } from './rings';
15
23
  import type { SessionSnapshot } from './session-state';
16
24
  import type {
17
25
  CaptureOptions,
@@ -88,6 +96,25 @@ const readStringFrom = (owner: unknown, key: string): string | undefined => {
88
96
  return typeof answer === 'string' ? answer : undefined;
89
97
  };
90
98
 
99
+ /**
100
+ * A `pageerror` payload, read defensively — never cast, and never assumed to be an `Error`.
101
+ *
102
+ * `readStringFrom`, the same reader the console handler uses, because the payload has the same
103
+ * problem: `message` and `stack` are an own property on one build and an accessor on another, and
104
+ * a schema parse cannot call an accessor. A page can also `throw 'a string'` or throw a frozen
105
+ * object with no `message` at all — both reach here, and both are recorded as SOMETHING having
106
+ * thrown, because an entry with a poor message is still the difference between "the island threw"
107
+ * and silence.
108
+ */
109
+ const readPageError = (payload: unknown, at: number): PageError => {
110
+ if (typeof payload === 'string') return pageErrorEntry({ message: payload, at });
111
+ return pageErrorEntry({
112
+ message: readStringFrom(payload, 'message') ?? '',
113
+ stack: readStringFrom(payload, 'stack'),
114
+ at,
115
+ });
116
+ };
117
+
91
118
  /**
92
119
  * The one failure `guard()` must NOT re-label, and the line is drawn at exactly one code.
93
120
  *
@@ -116,17 +143,25 @@ export interface CdpTargetInit {
116
143
  readonly ringCapacity?: number | undefined;
117
144
  }
118
145
 
146
+ /**
147
+ * Everything `arm()` writes into. Named rather than positional: three rings of near-identical
148
+ * type plus a latch is a call site nobody can read, and swapping two of them is a mistake the
149
+ * compiler cannot catch.
150
+ */
151
+ interface CdpSinks {
152
+ readonly network: NetworkRing;
153
+ readonly console: ConsoleRing;
154
+ readonly pageErrors: PageErrorRing;
155
+ readonly crashed: { value: string | undefined };
156
+ }
157
+
119
158
  /**
120
159
  * Interception is armed BEFORE the first navigation and refuses at the request, not after the
121
160
  * response — an `allowHosts` that reported afterwards would be a log line about bytes that
122
161
  * already left the container.
123
162
  */
124
- async function arm(
125
- init: CdpTargetInit,
126
- network: ReturnType<typeof createRing<NetworkEntry>>,
127
- console_: ReturnType<typeof createRing<ConsoleLine>>,
128
- crashed: { value: string | undefined },
129
- ): Promise<void> {
163
+ async function arm(init: CdpTargetInit, sinks: CdpSinks): Promise<void> {
164
+ const { network, console: console_, pageErrors, crashed } = sinks;
130
165
  await init.page.setRequestInterception(true);
131
166
  init.page.on('request', (payload) => {
132
167
  const request = asRequest(payload);
@@ -154,6 +189,23 @@ async function arm(
154
189
  at: init.clock.now().getTime(),
155
190
  });
156
191
  });
192
+ /**
193
+ * The page threw and nothing caught it. THE gap this ring closes: a screenshot of an island
194
+ * that threw during hydration is a picture of the server-rendered markup, indistinguishable
195
+ * from a page that worked — and `console` does not carry it, because throwing calls no console
196
+ * method. Subscribed here, beside the others, so a target is observing before its first
197
+ * navigation: an exception raised during load has no second chance to be recorded.
198
+ *
199
+ * NOT the same event as `error` below, and the difference is the whole reason this is a
200
+ * separate handler: puppeteer's `pageerror` is "an uncaught exception happens within the page"
201
+ * and its `error` is "the page crashes" (`PageEvent.PageError` / `PageEvent.Error`). One is the
202
+ * app being broken and the session is fine; the other is the tab being gone. Recording a
203
+ * `pageerror` into `crashed` would make every scrape of a page with one bad island answer
204
+ * X_SCRAPE_PAGE_CRASHED — a code registered `terminal` — for a page still perfectly usable.
205
+ */
206
+ init.page.on('pageerror', (payload) => {
207
+ pageErrors.push(readPageError(payload, init.clock.now().getTime()));
208
+ });
157
209
  // A renderer that dies must be a CODE, not a hang: every later call answers X_SCRAPE_PAGE_CRASHED
158
210
  // instead of waiting out its own timeout against a tab that is gone.
159
211
  init.page.on('error', (payload) => {
@@ -164,8 +216,9 @@ async function arm(
164
216
  export async function cdpTarget(init: CdpTargetInit): Promise<ScrapeTarget> {
165
217
  const console_ = createRing<ConsoleLine>(init.ringCapacity);
166
218
  const network = createRing<NetworkEntry>(init.ringCapacity);
219
+ const pageErrors = createRing<PageError>(init.ringCapacity);
167
220
  const crashed: { value: string | undefined } = { value: undefined };
168
- await arm(init, network, console_, crashed);
221
+ await arm(init, { network, console: console_, pageErrors, crashed });
169
222
  let pendingStorage: SessionSnapshot | undefined;
170
223
 
171
224
  const originOf = (url: string): string => {
@@ -233,6 +286,9 @@ export async function cdpTarget(init: CdpTargetInit): Promise<ScrapeTarget> {
233
286
  driver: CDP_DRIVER,
234
287
  console: console_,
235
288
  network,
289
+ // Shared with every frame target below, through the spread: an exception is the PAGE's, and a
290
+ // per-frame ring would hide a throw from an iframe behind whichever handle the caller held.
291
+ pageErrors,
236
292
  url: () => init.page.url(),
237
293
  goto: (url: string, options: GotoOptions) =>
238
294
  guard('goto', async () => {
@@ -106,6 +106,28 @@ export const wedged = (what: string, idleMs: number): ScrapeError =>
106
106
  meta: { what, idleMs },
107
107
  });
108
108
 
109
+ /**
110
+ * The watchdog's SECOND way of ending a run, and its OWN code rather than `X_SCRAPE_WEDGED`.
111
+ *
112
+ * An exact `cause:` cannot rescue a wrong title. A reader who hits this runs
113
+ * `x errors explain X_SCRAPE_WEDGED`, reads "the browser stopped answering and was killed", and
114
+ * goes to investigate a page that is fine — the browser never stopped answering, the guard's own
115
+ * loop died on code the DEFINITION supplied. Two events with two different subsystems and two
116
+ * opposite fixes are two codes, and the classifications differ with them: a wedge is retryable,
117
+ * this is terminal (`errors.ts`).
118
+ *
119
+ * Reachable through the `ScrapeClock` seam and the driver's `kill()`, both of which a third party
120
+ * writes. Leaving the loop dead and quiet is the alternative, and that is incident #1 with no
121
+ * guard armed at all.
122
+ */
123
+ export const watchdogStopped = (what: string, thrown: unknown): ScrapeError =>
124
+ new ScrapeError({
125
+ code: 'X_SCRAPE_WATCHDOG_STOPPED',
126
+ cause: `the wedge watchdog for ${what} stopped measuring browser activity: ${renderThrowable(thrown)}`,
127
+ fix: 'drop the custom clock: from the scrape() definition — systemScrapeClock is the only ScrapeClock whose sleep() cannot reject — then re-run',
128
+ meta: { what },
129
+ });
130
+
109
131
  export const pageCrashed = (url: string): ScrapeError =>
110
132
  new ScrapeError({
111
133
  code: 'X_SCRAPE_PAGE_CRASHED',
package/src/errors.ts CHANGED
@@ -22,6 +22,7 @@ export const SCRAPE_OWNED_ERROR_CODES = [
22
22
  'X_SCRAPE_NOT_ACTIONABLE',
23
23
  'X_SCRAPE_TIMEOUT',
24
24
  'X_SCRAPE_WEDGED',
25
+ 'X_SCRAPE_WATCHDOG_STOPPED',
25
26
  'X_SCRAPE_PAGE_CRASHED',
26
27
  'X_SCRAPE_OUTPUT_INVALID',
27
28
  'X_SCRAPE_YIELD_COLLAPSED',
@@ -68,6 +69,7 @@ export const SCRAPE_ERROR_TITLES: Readonly<Record<ScrapeOwnedErrorCode, string>>
68
69
  X_SCRAPE_NOT_ACTIONABLE: 'the element is present and cannot be acted on',
69
70
  X_SCRAPE_TIMEOUT: 'the step exceeded its wall-clock budget',
70
71
  X_SCRAPE_WEDGED: 'the browser stopped answering and was killed',
72
+ X_SCRAPE_WATCHDOG_STOPPED: 'the wedge watchdog stopped measuring and the run was ended',
71
73
  X_SCRAPE_PAGE_CRASHED: 'the renderer process died',
72
74
  X_SCRAPE_OUTPUT_INVALID: 'the extracted rows do not match the extract schema',
73
75
  X_SCRAPE_YIELD_COLLAPSED: 'the run succeeded and returned far too little',
@@ -138,6 +140,14 @@ export const SCRAPE_ERROR_RETRY = {
138
140
  // A declaration error, raised by `scrape()` before any attempt exists — there is no run to
139
141
  // retry, and the same definition would refuse identically forever.
140
142
  X_SCRAPE_YIELD_HISTORY_MISSING: 'terminal',
143
+ // TERMINAL where its sibling `X_SCRAPE_WEDGED` is retryable, and the difference IS the reason
144
+ // the two codes are separate. A wedge is the site or the browser being slow — the definition of
145
+ // "run it again and it may go differently". This is the guard's own loop dying on code the
146
+ // DEFINITION supplied: a `ScrapeClock` an app wrote, reached identically on attempt 2. Retrying
147
+ // launches a real browser five times to die at the first poll, and on an authenticated target
148
+ // that is five arrivals at a login for no chance of a different answer — the rule this whole
149
+ // table is written to, stated at the top of it. The fix is an edit, so a human decides.
150
+ X_SCRAPE_WATCHDOG_STOPPED: 'terminal',
141
151
  X_SCRAPE_ROBOTS_DISALLOWED: 'terminal',
142
152
  X_SCRAPE_FIXTURE_MISSING: 'terminal',
143
153
  X_SCRAPE_FIXTURE_STALE: 'terminal',
@@ -13,7 +13,7 @@ import type { InterceptRules } from './intercept';
13
13
  import { interceptVerdict, refusalEntry } from './intercept';
14
14
  import type { PageRecording } from './recording';
15
15
  import { splitDownload } from './recording';
16
- import type { ConsoleLine, NetworkEntry } from './rings';
16
+ import type { ConsoleLine, NetworkEntry, PageError } from './rings';
17
17
  import { createRing } from './rings';
18
18
  import type { SessionSnapshot } from './session-state';
19
19
  import { EMPTY_SESSION } from './session-state';
@@ -74,6 +74,12 @@ const keyOf = (selector: string, element: ElementSnapshot | undefined): string =
74
74
  export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
75
75
  const consoleRing = createRing<ConsoleLine>();
76
76
  const networkRing = createRing<NetworkEntry>();
77
+ // Built and never pushed to, deliberately: this target parses markup and executes none of it, so
78
+ // there is no uncaught exception for it to have. It ANSWERS rather than omitting the ring —
79
+ // `page.pageErrors()` returning `[]` here is the honest "nothing threw, and nothing could",
80
+ // where a missing ring would be a page method that throws on two of the three drivers.
81
+ // `driver-parity.test.ts` pins the divergence, beside the box/hit-target one it already carries.
82
+ const pageErrorRing = createRing<PageError>();
77
83
  const overlay = new Map<string, string>();
78
84
  let page: PageRecording = init.start ?? EMPTY;
79
85
  let armed: string | undefined;
@@ -164,6 +170,7 @@ export function htmlTarget(init: HtmlTargetInit): ScrapeTarget {
164
170
  driver: init.driver,
165
171
  console: consoleRing,
166
172
  network: networkRing,
173
+ pageErrors: pageErrorRing,
167
174
  url: () => page.url,
168
175
  goto: (url: string, _options: GotoOptions): Promise<void> => navigate(url),
169
176
  content: (): Promise<string> => {
package/src/index.ts CHANGED
@@ -137,10 +137,18 @@ export type {
137
137
  ConsoleRing,
138
138
  NetworkEntry,
139
139
  NetworkRing,
140
+ PageError,
141
+ PageErrorRing,
140
142
  ResourceType,
141
143
  Ring,
142
144
  } from './rings';
143
- export { createRing, DEFAULT_RING_CAPACITY, RESOURCE_TYPES } from './rings';
145
+ export {
146
+ createRing,
147
+ DEFAULT_RING_CAPACITY,
148
+ MAX_PAGE_ERROR_CHARS,
149
+ pageErrorEntry,
150
+ RESOURCE_TYPES,
151
+ } from './rings';
144
152
  export type { RobotsFetch, RobotsGate, RobotsGateInit, RobotsPolicy, RobotsRules } from './robots';
145
153
  export { createRobotsGate, DEFAULT_ROBOTS_AGENT, parseRobots, robotsAllows } from './robots';
146
154
  export type { RobotsFetchInit } from './robots-fetch';
@@ -192,6 +192,12 @@ export function pageOverTarget(target: ScrapeTarget, ctx: PageContext): ScrapePa
192
192
  };
193
193
  return {
194
194
  ...frame,
195
+ // The DOCUMENT's URL is asked of the target, never of the frame's cached `lastUrl`. `ScrapeFrame`
196
+ // resolves its target asynchronously and `url()` is synchronous, so a child frame can only ever
197
+ // answer from a cache refreshed on its last wait — but the page HOLDS its target, so it has no
198
+ // such excuse. Spreading `...frame` without this override made `page.url()` answer the seed
199
+ // (`about:blank`) after every `goto`, which is what `x shot` reported as `finalUrl`.
200
+ url: () => target.url(),
195
201
  async goto(url, options): Promise<void> {
196
202
  await guardNavigation(url, ctx);
197
203
  await ctx.pace?.(ctx.signal);
@@ -212,7 +218,9 @@ export function pageOverTarget(target: ScrapeTarget, ctx: PageContext): ScrapePa
212
218
  cookies: (): Promise<readonly ScrapeCookie[]> => target.cookies(),
213
219
  session: () => target.session(),
214
220
  console: () => target.console.entries(),
221
+ pageErrors: () => target.pageErrors.entries(),
215
222
  network: () => target.network.entries(),
216
223
  networkDropped: () => target.network.dropped,
224
+ pageErrorsDropped: () => target.pageErrors.dropped,
217
225
  };
218
226
  }
package/src/page.ts CHANGED
@@ -8,7 +8,7 @@
8
8
 
9
9
  import type { Secret } from '@ultimat3/core';
10
10
  import type { ActionabilityState } from './actionability';
11
- import type { ConsoleLine, NetworkEntry } from './rings';
11
+ import type { ConsoleLine, NetworkEntry, PageError } from './rings';
12
12
  import type { SessionSnapshot } from './session-state';
13
13
  import type { ElementSnapshot, ScrapeCookie, ScrapeDownloadFile } from './target';
14
14
 
@@ -93,6 +93,13 @@ export interface ScrapePage extends ScrapeFrame {
93
93
  session(): Promise<SessionSnapshot>;
94
94
  /** The bounded tail. Bounded because a long run's full history is an OOM, not a log. */
95
95
  console(): readonly ConsoleLine[];
96
+ /**
97
+ * The uncaught exceptions the page threw, which `console()` does NOT carry: an island that
98
+ * throws calls no console method, so a scrape reading console alone sees a page that looks
99
+ * silent and is broken. Empty on a driver with no JS engine — the offline drivers parse HTML
100
+ * and never execute it, so nothing there can throw.
101
+ */
102
+ pageErrors(): readonly PageError[];
96
103
  network(): readonly NetworkEntry[];
97
104
  /**
98
105
  * How many entries the bound above threw away. It is the same honesty `Ring.dropped` carries:
@@ -100,4 +107,10 @@ export interface ScrapePage extends ScrapeFrame {
100
107
  * is non-zero, and a scrape that blocked 5,000 images otherwise reports 200 with no hint.
101
108
  */
102
109
  networkDropped(): number;
110
+ /**
111
+ * The same honesty for the errors, and it is the count a verdict gates on: an island throwing
112
+ * inside a `requestAnimationFrame` produces thousands, and "3 page errors" read off a bounded
113
+ * tail of 200 would be a number a reader trusts and should not.
114
+ */
115
+ pageErrorsDropped(): number;
103
116
  }
package/src/rings.ts CHANGED
@@ -1,6 +1,6 @@
1
- // Bounded console and network history. BOUNDED is the whole point: a scrape of ten thousand
2
- // pages that kept every console line and every request holds the run's entire browsing history in
3
- // the worker's heap, and the incident is an OOM two hours in rather than a scraper bug.
1
+ // Bounded console, page-error and network history. BOUNDED is the whole point: a scrape of ten
2
+ // thousand pages that kept every console line and every request holds the run's entire browsing
3
+ // history in the worker's heap, and the incident is an OOM two hours in rather than a scraper bug.
4
4
 
5
5
  export interface ConsoleLine {
6
6
  readonly level: 'log' | 'info' | 'warn' | 'error' | 'debug';
@@ -19,6 +19,55 @@ export interface NetworkEntry {
19
19
  readonly refused?: 'blocked' | 'host' | 'robots' | undefined;
20
20
  }
21
21
 
22
+ /**
23
+ * An uncaught exception the PAGE threw — the half a picture cannot carry.
24
+ *
25
+ * Its own entry type, and not a `ConsoleLine` with `level: 'error'`, because it is a different
26
+ * event: a page that threw called no console method, so folding one into the console stream
27
+ * reports a `console.error` that never happened — and `ConsoleLine` has nowhere to keep the stack,
28
+ * which is the field that says WHICH island threw.
29
+ */
30
+ export interface PageError {
31
+ /** The exception's message. `''` when the payload carried none — never a fabricated one. */
32
+ readonly message: string;
33
+ /**
34
+ * The JS stack, when the exception carried one. Absent means the payload had none, never that
35
+ * the throw had no origin: a `throw 'a string'` in the page reaches here with a message alone.
36
+ */
37
+ readonly stack?: string | undefined;
38
+ readonly at: number;
39
+ }
40
+
41
+ /**
42
+ * The per-entry cap, which the ring itself cannot give: a ring bounds the COUNT, and one
43
+ * `RangeError: Maximum call stack size exceeded` carries thousands of frames — 200 of those is
44
+ * megabytes held per page, which is the same OOM this file exists to prevent, one level down.
45
+ */
46
+ export const MAX_PAGE_ERROR_CHARS = 4_000;
47
+
48
+ const clamped = (text: string): string =>
49
+ text.length <= MAX_PAGE_ERROR_CHARS ? text : `${text.slice(0, MAX_PAGE_ERROR_CHARS - 1)}…`;
50
+
51
+ /**
52
+ * The ONE way a `PageError` is built. A driver that assembled the object literal itself would be
53
+ * a driver whose stacks are unbounded, and the truncation would then be a rule per driver rather
54
+ * than a property of the type.
55
+ */
56
+ export function pageErrorEntry(input: {
57
+ readonly message: string;
58
+ readonly stack?: string | undefined;
59
+ readonly at: number;
60
+ }): PageError {
61
+ // An empty stack is ABSENT, not empty: `stack: ''` prints as a blank stack, which reads as
62
+ // "the exception had no origin" rather than "the payload did not carry one".
63
+ const stack = input.stack === undefined || input.stack === '' ? undefined : clamped(input.stack);
64
+ return {
65
+ message: clamped(input.message),
66
+ ...(stack === undefined ? {} : { stack }),
67
+ at: input.at,
68
+ };
69
+ }
70
+
22
71
  export const RESOURCE_TYPES = [
23
72
  'document',
24
73
  'stylesheet',
@@ -48,6 +97,7 @@ export interface Ring<T> {
48
97
 
49
98
  export type ConsoleRing = Ring<ConsoleLine>;
50
99
  export type NetworkRing = Ring<NetworkEntry>;
100
+ export type PageErrorRing = Ring<PageError>;
51
101
 
52
102
  export function createRing<T>(capacity: number = DEFAULT_RING_CAPACITY): Ring<T> {
53
103
  const items: T[] = [];
package/src/scrape.ts CHANGED
@@ -1,5 +1,6 @@
1
- // `scrape()` — a browser run, declared as a `job` and NOT as a ninth primitive. The rule's fourth
2
- // instance after `llm()` (an action factory) and `backfill()` (a job factory).
1
+ // `scrape()` — a browser run, declared as a `job` and NOT as a ninth primitive. One row of
2
+ // `PRIMITIVE_FACTORIES` in `@ultimat3/core`, which is the derived list of every factory that
3
+ // ships; an ordinal written here would be wrong the moment the next one lands, and was.
3
4
  //
4
5
  // It is a job by every field of the definition, not by analogy: a scrape has an input schema, a
5
6
  // tenant, a retry policy, a timeout, a concurrency cap, a queue, and — decisively — a REQUIRED
package/src/target.ts CHANGED
@@ -3,7 +3,7 @@
3
3
  // lets `page-over-target.ts` be the ONE implementation of `ScrapePage` for the real browser, the
4
4
  // fixture replayer and the fake alike.
5
5
 
6
- import type { ConsoleRing, NetworkRing } from './rings';
6
+ import type { ConsoleRing, NetworkRing, PageErrorRing } from './rings';
7
7
  import type { SessionSnapshot } from './session-state';
8
8
 
9
9
  /**
@@ -96,6 +96,15 @@ export interface ScrapeTarget {
96
96
  readonly driver: string;
97
97
  readonly console: ConsoleRing;
98
98
  readonly network: NetworkRing;
99
+ /**
100
+ * REQUIRED, and empty is a legitimate answer. A driver with no JS engine has no uncaught
101
+ * exception to report and answers an empty ring; an optional member would instead let a driver
102
+ * be silent about errors it CAN see, which is precisely the blind spot this ring closes — and
103
+ * `ScrapePage.pageErrors()` would then be a method that answers nothing on an unknown subset of
104
+ * drivers. A third-party driver added before 2026-08-21 gets a type error naming this field,
105
+ * which is the whole enforcement.
106
+ */
107
+ readonly pageErrors: PageErrorRing;
99
108
  url(): string;
100
109
  goto(url: string, options: GotoOptions): Promise<void>;
101
110
  /** Serialised HTML of THIS target — the document for a page, the subtree for a frame. */
package/src/watchdog.ts CHANGED
@@ -11,7 +11,7 @@
11
11
  // graceful-quit CEILING on shutdown, past which the same kill runs.
12
12
 
13
13
  import type { ScrapeClock } from './clock';
14
- import { wedged } from './error-throws';
14
+ import { watchdogStopped, wedged } from './error-throws';
15
15
 
16
16
  export const DEFAULT_IDLE_MS = 120_000;
17
17
  export const DEFAULT_GRACE_MS = 5_000;
@@ -41,7 +41,10 @@ export interface WedgeGuard {
41
41
  touch(): void;
42
42
  /** Graceful quit under the ceiling, then kill. Idempotent, and never throws. */
43
43
  shutdown(): Promise<void>;
44
- /** True once the watchdog fired — a run that ends after this ended because of it. */
44
+ /**
45
+ * True once the watchdog ended the run — the idle budget passed, or the guard's own loop died
46
+ * and stopped measuring. Either way a run that ends after this ended because of the watchdog.
47
+ */
45
48
  readonly fired: boolean;
46
49
  }
47
50
 
@@ -50,7 +53,13 @@ export function createWedgeGuard(init: WedgeGuardInit): WedgeGuard {
50
53
  const graceMs = init.graceMs ?? DEFAULT_GRACE_MS;
51
54
  const controller = new AbortController();
52
55
  let lastTouch = init.clock.monotonic();
56
+ // TWO latches, not one. `stopped` ends the watch loop; `shuttingDown` guards the teardown. They
57
+ // were the same flag, so a fire — which sets `stopped` — made `shutdown()` return on its first
58
+ // line and NEVER call `quit()`. `localBrowser()` hid it, because the fire's `kill()` still
59
+ // reaches a pid; `remoteBrowser()` is `driver-cdp.ts`'s primary path, `browser.process()` is
60
+ // `null` there, and `browser.close()` is then the only thing that ends the paid remote session.
53
61
  let stopped = false;
62
+ let shuttingDown = false;
54
63
  let fired = false;
55
64
 
56
65
  const watch = async (): Promise<void> => {
@@ -67,7 +76,17 @@ export function createWedgeGuard(init: WedgeGuardInit): WedgeGuard {
67
76
  controller.abort(wedged(init.what, idleMs));
68
77
  }
69
78
  };
70
- void watch();
79
+ // Never floating. `ScrapeClock` is a seam an app implements and `init.kill()` comes from a
80
+ // driver, so this loop can reject on somebody else's code — and a bare `void` turned that into
81
+ // an unhandled rejection, a process-level event belonging to no run, with the guard silently
82
+ // dead behind it. A guard that stopped measuring is incident #1 with nothing armed, so the run
83
+ // is ended with an instruction instead of left unwatched.
84
+ void watch().catch((thrown: unknown) => {
85
+ if (stopped) return;
86
+ stopped = true;
87
+ fired = true;
88
+ controller.abort(watchdogStopped(init.what, thrown));
89
+ });
71
90
 
72
91
  return {
73
92
  signal: controller.signal,
@@ -78,7 +97,8 @@ export function createWedgeGuard(init: WedgeGuardInit): WedgeGuard {
78
97
  lastTouch = init.clock.monotonic();
79
98
  },
80
99
  async shutdown(): Promise<void> {
81
- if (stopped) return;
100
+ if (shuttingDown) return;
101
+ shuttingDown = true;
82
102
  stopped = true;
83
103
  let quit = false;
84
104
  // A ceiling, not a hope: whichever finishes first wins, and if it is the clock the process
@@ -92,7 +112,11 @@ export function createWedgeGuard(init: WedgeGuardInit): WedgeGuard {
92
112
  quit = false;
93
113
  },
94
114
  ),
95
- init.clock.sleep(graceMs),
115
+ // The ceiling's own sleep is caught for the same reason the quit is: this runs in
116
+ // `runScrape`'s `finally`, `close()` is documented never to throw, and a clock that
117
+ // rejected here would replace the run's real failure with a teardown one. A ceiling that
118
+ // cannot be timed has already expired, which lands on `kill()` — the safe direction.
119
+ init.clock.sleep(graceMs).catch(() => undefined),
96
120
  ]);
97
121
  if (!quit) init.kill();
98
122
  },