@ultimat3/scraping 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,258 @@
1
+ // One constructor per failure mode, so a throw site is one call and every `cause`/`fix` pair for
2
+ // a code is written once. Nothing here interpolates an `unknown` into a cause: an exception from
3
+ // a browser is rendered through core's `renderThrowable`, which is the rule `bun run error-render`
4
+ // enforces.
5
+
6
+ import { renderThrowable, UltimateError } from '@ultimat3/core';
7
+ import { ScrapeError } from './errors';
8
+
9
+ /**
10
+ * Two shapes of the same failure, one code. `name` is the driver a definition ASKED for; `scrape`
11
+ * is passed when there is no driver at all to name — and that is the reachable case, so the cause
12
+ * has to read as one. It said `no scrape driver named "orders.daily" is installed`, naming the
13
+ * scrape as a driver, which sends its reader hunting for a driver nobody ever declared.
14
+ */
15
+ export const driverUnknown = (
16
+ name: string | undefined,
17
+ installed: readonly string[],
18
+ scrape?: string,
19
+ ): ScrapeError =>
20
+ new ScrapeError({
21
+ code: 'X_SCRAPE_DRIVER_UNKNOWN',
22
+ cause:
23
+ scrape === undefined
24
+ ? `no scrape driver named "${name ?? 'none'}" is installed; installed: ${installed.join(', ') || 'none'}`
25
+ : `scrape "${scrape}" has no browser driver: nothing called setScrapeDriver() and the definition declares no driver; installed: ${installed.join(', ') || 'none'}`,
26
+ fix: 'call setScrapeDriver(localBrowser()) at boot, or pass driver: fakeBrowser() on the scrape() definition',
27
+ meta: { driver: name ?? 'none', ...(scrape === undefined ? {} : { scrape }) },
28
+ });
29
+
30
+ export const cdpAttachFailed = (cdpUrl: string, thrown: unknown): ScrapeError =>
31
+ new ScrapeError({
32
+ code: 'X_SCRAPE_CDP_ATTACH_FAILED',
33
+ cause: `the CDP endpoint ${cdpUrl} refused the attach: ${renderThrowable(thrown)}`,
34
+ fix: 'curl "$CDP_URL/json/version" to confirm the endpoint answers, then pass that webSocketDebuggerUrl to remoteBrowser({ cdpUrl })',
35
+ meta: { cdpUrl },
36
+ });
37
+
38
+ export const browserUnreachable = (driver: string, thrown: unknown): ScrapeError =>
39
+ new ScrapeError({
40
+ code: 'X_SCRAPE_BROWSER_UNREACHABLE',
41
+ cause: `the ${driver} browser stopped answering: ${renderThrowable(thrown)}`,
42
+ fix: 'raise watchdog: { idleMs } on the scrape() definition if the site is genuinely slow, otherwise re-run once the browser host is back',
43
+ meta: { driver },
44
+ });
45
+
46
+ export const profileLocked = (profileDir: string): ScrapeError =>
47
+ new ScrapeError({
48
+ code: 'X_SCRAPE_PROFILE_LOCKED',
49
+ cause: `another browser process holds the profile at ${profileDir}`,
50
+ fix: `rm -f ${profileDir}/SingletonLock once no browser is using it, or give this run its own localBrowser({ profileDir })`,
51
+ meta: { profileDir },
52
+ });
53
+
54
+ export const hostBlocked = (url: string, allowed: readonly string[]): ScrapeError =>
55
+ new ScrapeError({
56
+ code: 'X_SCRAPE_HOST_BLOCKED',
57
+ cause: `the page requested ${url}, and allowHosts lists ${allowed.join(', ') || 'nothing'}`,
58
+ fix: `add the host to allowHosts on the scrape() definition — allowHosts: [${allowed.map((host) => `'${host}'`).join(', ')}, '<the host above>']`,
59
+ meta: { url, allowed },
60
+ });
61
+
62
+ export const selectorMissing = (selector: string, url: string, waitedMs: number): ScrapeError =>
63
+ new ScrapeError({
64
+ code: 'X_SCRAPE_SELECTOR_MISSING',
65
+ cause: `"${selector}" never appeared on ${url} within ${String(waitedMs)}ms`,
66
+ fix: "open the run's page.html artifact and re-derive the selector, or raise the timeout on this waitFor({ timeout })",
67
+ meta: { selector, url, waitedMs },
68
+ });
69
+
70
+ export const notActionable = (selector: string, reason: string, waitedMs: number): ScrapeError =>
71
+ new ScrapeError({
72
+ code: 'X_SCRAPE_NOT_ACTIONABLE',
73
+ cause: `"${selector}" is present and ${reason} after ${String(waitedMs)}ms`,
74
+ fix: 'wait for the state that unblocks it — page.waitFor(selector, { state: "enabled" }) — or dismiss whatever overlays it before the click',
75
+ meta: { selector, reason, waitedMs },
76
+ });
77
+
78
+ export const scrapeTimeout = (what: string, ms: number): ScrapeError =>
79
+ new ScrapeError({
80
+ code: 'X_SCRAPE_TIMEOUT',
81
+ cause: `${what} exceeded its ${String(ms)}ms budget`,
82
+ fix: 'raise timeout: on the scrape() definition, or split the pass so each step.run stays inside one budget',
83
+ meta: { what, ms },
84
+ });
85
+
86
+ export const wedged = (what: string, idleMs: number): ScrapeError =>
87
+ new ScrapeError({
88
+ code: 'X_SCRAPE_WEDGED',
89
+ cause: `${what} produced no browser activity for ${String(idleMs)}ms, so the process was killed`,
90
+ fix: 'raise watchdog: { idleMs } on the scrape() definition if the site is genuinely this slow, otherwise re-run — the browser was killed and its session is gone',
91
+ meta: { what, idleMs },
92
+ });
93
+
94
+ export const pageCrashed = (url: string): ScrapeError =>
95
+ new ScrapeError({
96
+ code: 'X_SCRAPE_PAGE_CRASHED',
97
+ cause: `the renderer for ${url} died mid-run, so this attempt's page state is gone`,
98
+ fix: 'lower concurrency: on the scrape() definition or raise the container memory limit in docker-compose.prod.yml — a crashed renderer is out of memory far more often than it is a bug',
99
+ meta: { url },
100
+ });
101
+
102
+ export const outputInvalid = (name: string, detail: string): ScrapeError =>
103
+ new ScrapeError({
104
+ code: 'X_SCRAPE_OUTPUT_INVALID',
105
+ cause: `scrape "${name}" produced rows its extract schema rejects: ${detail}`,
106
+ fix: "align the extract schema with the page — the run's page.html artifact holds the markup the rows came from",
107
+ meta: { scrape: name },
108
+ });
109
+
110
+ export const downloadTimeout = (ms: number, url: string): ScrapeError =>
111
+ new ScrapeError({
112
+ code: 'X_SCRAPE_DOWNLOAD_TIMEOUT',
113
+ cause: `no download landed within ${String(ms)}ms of the trigger on ${url}`,
114
+ fix: 'raise the timeout on page.download({ timeout }), or confirm the trigger is the element that starts the download',
115
+ meta: { ms, url },
116
+ });
117
+
118
+ /**
119
+ * The silent-green alarm. A scraper that succeeds and returns nothing stays green for weeks, and
120
+ * every field of this cause is there so the first read answers "was it always this low, or did it
121
+ * fall off a cliff today?" without opening a dashboard.
122
+ */
123
+ export const yieldCollapsed = (input: {
124
+ readonly scrape: string;
125
+ readonly rows: number;
126
+ readonly reason: 'min-rows' | 'drop';
127
+ readonly minRows?: number | undefined;
128
+ readonly baseline?: number | undefined;
129
+ readonly maxDrop?: number | undefined;
130
+ }): ScrapeError =>
131
+ new ScrapeError({
132
+ code: 'X_SCRAPE_YIELD_COLLAPSED',
133
+ cause:
134
+ input.reason === 'min-rows'
135
+ ? `scrape "${input.scrape}" returned ${String(input.rows)} rows and declares expect.minRows ${String(input.minRows)}`
136
+ : `scrape "${input.scrape}" returned ${String(input.rows)} rows against a trailing median of ${String(input.baseline)}, past the ${String(Math.round((input.maxDrop ?? 0) * 100))}% drop expect.maxDrop allows`,
137
+ fix: "open the run's page.html artifact and compare it with the extract selectors — a collapse is the page changing far more often than the data changing",
138
+ meta: {
139
+ scrape: input.scrape,
140
+ rows: input.rows,
141
+ reason: input.reason,
142
+ minRows: input.minRows,
143
+ baseline: input.baseline,
144
+ maxDrop: input.maxDrop,
145
+ },
146
+ });
147
+
148
+ export const robotsDisallowed = (url: string, agent: string): ScrapeError =>
149
+ new ScrapeError({
150
+ code: 'X_SCRAPE_ROBOTS_DISALLOWED',
151
+ cause: `robots.txt disallows ${url} for user-agent "${agent}"`,
152
+ fix: "declare robots: { ignore: '<the written reason this run is permitted>' } on the scrape() definition, or scrape a path robots.txt allows",
153
+ meta: { url, agent },
154
+ });
155
+
156
+ export const fixtureMissing = (url: string, dir: string): ScrapeError =>
157
+ new ScrapeError({
158
+ code: 'X_SCRAPE_FIXTURE_MISSING',
159
+ cause: `the fixture driver has no recording of ${url} under ${dir}, and an offline driver never reaches the network`,
160
+ fix: `add ${dir}/<the recording file> with { "url": "${url}", "html": "…" }, or point fixtureBrowser() at the directory that already holds it`,
161
+ meta: { url, dir },
162
+ });
163
+
164
+ export const fixtureStale = (url: string, ageMs: number, maxAgeMs: number): ScrapeError =>
165
+ new ScrapeError({
166
+ code: 'X_SCRAPE_FIXTURE_STALE',
167
+ cause: `the recording of ${url} is ${String(Math.round(ageMs / 86_400_000))} days old and fixtureBrowser declares maxAge ${String(Math.round(maxAgeMs / 86_400_000))} days`,
168
+ fix: 're-record the fixture directory, or raise maxAge on fixtureBrowser({ maxAge }) with the reason an old recording still proves something',
169
+ meta: { url, ageMs, maxAgeMs },
170
+ });
171
+
172
+ export const remoteRequired = (driver: string): ScrapeError =>
173
+ new ScrapeError({
174
+ code: 'X_SCRAPE_REMOTE_REQUIRED',
175
+ cause: `the ${driver} driver attaches to a browser somebody else started and was given no cdpUrl`,
176
+ fix: 'pass remoteBrowser({ cdpUrl: env.SCRAPE_CDP_URL }), or use localBrowser({ launcher }) to start one in this container',
177
+ meta: { driver },
178
+ });
179
+
180
+ export const recoverRefused = (name: string, reason: string): ScrapeError =>
181
+ new ScrapeError({
182
+ code: 'X_SCRAPE_RECOVER_REFUSED',
183
+ cause: `the recover hook on scrape "${name}" declined this failure: ${reason}`,
184
+ fix: "widen recover() to handle this failure, or drop recover: and let the job's retry policy own it",
185
+ meta: { scrape: name, reason },
186
+ });
187
+
188
+ export const secretExposed = (artifact: string, url: string): ScrapeError =>
189
+ new ScrapeError({
190
+ code: 'X_SCRAPE_SECRET_EXPOSED',
191
+ cause: `a ${artifact} of ${url} was requested after a secret was typed into this page, and pixels cannot be redacted afterwards`,
192
+ fix: 'call artifact.html() instead — page HTML is redacted by value and password fields are blanked — or take the capture before the secret is typed',
193
+ meta: { artifact, url },
194
+ });
195
+
196
+ /** A non-2xx from the HTTP leg. 4xx below 429 is terminal; everything else may be tried again. */
197
+ export const httpFailed = (url: string, status: number, body: string): ScrapeError =>
198
+ new ScrapeError({
199
+ code: 'X_SCRAPE_HTTP_FAILED',
200
+ cause: `${url} answered ${String(status)}: ${body}`,
201
+ fix: 'confirm the endpoint the browser leg calls is the one this request names — page.network() lists every URL the page actually fetched',
202
+ meta: { url, status },
203
+ retry: status >= 400 && status < 500 && status !== 429 ? 'terminal' : 'retryable',
204
+ });
205
+
206
+ /**
207
+ * TERMINAL, and the retry table cannot be talked out of it. A site that locks an account after
208
+ * three wrong attempts turns a retrying framework into the thing that destroys the user's
209
+ * account, so this failure ends the run — no backoff, no recovery hook, no second attempt.
210
+ */
211
+ export const authFailed = (scrape: string, detail: string): ScrapeError =>
212
+ new ScrapeError({
213
+ code: 'X_SCRAPE_AUTH_FAILED',
214
+ cause: `scrape "${scrape}" was refused by the site's login: ${detail}`,
215
+ fix: 'correct the credential in .env.local for the name listed in secrets: on the scrape() definition — this run will NOT be retried, deliberately: a second wrong attempt is how an account gets locked',
216
+ meta: { scrape },
217
+ });
218
+
219
+ export const sessionExpired = (scrape: string, key: string): ScrapeError =>
220
+ new ScrapeError({
221
+ code: 'X_SCRAPE_SESSION_EXPIRED',
222
+ cause: `the stored session ${key} for scrape "${scrape}" failed its validate() probe and the definition declares no auth.login`,
223
+ fix: `add auth: { login } to scrape("${scrape}") so an expired session can be replaced, or drop auth.validate and let the body handle the logged-out page`,
224
+ meta: { scrape, key },
225
+ });
226
+
227
+ export const promptUnanswered = (scrape: string, label: string): ScrapeError =>
228
+ new ScrapeError({
229
+ code: 'X_SCRAPE_PROMPT_UNANSWERED',
230
+ cause: `scrape "${scrape}" asked for "${label}" and no prompt handler was declared`,
231
+ fix: `add prompt: async ({ label }) => await otpFor(label) to scrape("${scrape}") — a code that arrives out of band needs somewhere to come from`,
232
+ meta: { scrape, label },
233
+ });
234
+
235
+ /**
236
+ * The identity is spent. Retryable — and the run BURNS the session before the retry, because a
237
+ * flagged profile stays flagged and reloading it re-trips the same block every time.
238
+ */
239
+ export const blocked = (scrape: string, url: string, detail: string): ScrapeError =>
240
+ new ScrapeError({
241
+ code: 'X_SCRAPE_BLOCKED',
242
+ cause: `scrape "${scrape}" was refused by ${url}: ${detail}`,
243
+ fix: 'the session is burned and the next attempt starts a new identity — lower rate: on the scrape() definition if this repeats',
244
+ meta: { scrape, url },
245
+ });
246
+
247
+ /**
248
+ * The honest stub, in the shape `packages/jobs/src/driver-redis.ts` uses: correct types so an app
249
+ * can be written against the seam, and one labelled throw so nobody discovers the gap from a
250
+ * silently-skipped recovery.
251
+ */
252
+ export const scrapeNotImplemented = (feature: string, fix: string): UltimateError =>
253
+ new UltimateError({
254
+ code: 'X_NOT_IMPLEMENTED',
255
+ cause: `${feature} is declared and not implemented in @ultimat3/scraping`,
256
+ fix,
257
+ meta: { feature },
258
+ });
package/src/errors.ts ADDED
@@ -0,0 +1,180 @@
1
+ // The X_* codes owned by @ultimat3/scraping, and — the half that carries the weight — their
2
+ // retry classification. A scraper's whole operational question is "was that the site being slow
3
+ // or the site being different?", so every code here is registered as `retryable` or `terminal`
4
+ // once, in this file, rather than re-decided by whichever `catch` saw it.
5
+
6
+ import type { ErrorRetry } from '@ultimat3/core';
7
+ import {
8
+ errorDocsUrl,
9
+ registerErrorCodes,
10
+ registerErrorRetry,
11
+ UltimateError,
12
+ } from '@ultimat3/core';
13
+
14
+ /** Codes this package declares and owns. */
15
+ export const SCRAPE_OWNED_ERROR_CODES = [
16
+ 'X_SCRAPE_DRIVER_UNKNOWN',
17
+ 'X_SCRAPE_CDP_ATTACH_FAILED',
18
+ 'X_SCRAPE_BROWSER_UNREACHABLE',
19
+ 'X_SCRAPE_PROFILE_LOCKED',
20
+ 'X_SCRAPE_HOST_BLOCKED',
21
+ 'X_SCRAPE_SELECTOR_MISSING',
22
+ 'X_SCRAPE_NOT_ACTIONABLE',
23
+ 'X_SCRAPE_TIMEOUT',
24
+ 'X_SCRAPE_WEDGED',
25
+ 'X_SCRAPE_PAGE_CRASHED',
26
+ 'X_SCRAPE_OUTPUT_INVALID',
27
+ 'X_SCRAPE_YIELD_COLLAPSED',
28
+ 'X_SCRAPE_DOWNLOAD_TIMEOUT',
29
+ 'X_SCRAPE_ROBOTS_DISALLOWED',
30
+ 'X_SCRAPE_FIXTURE_MISSING',
31
+ 'X_SCRAPE_FIXTURE_STALE',
32
+ 'X_SCRAPE_REMOTE_REQUIRED',
33
+ 'X_SCRAPE_RECOVER_REFUSED',
34
+ 'X_SCRAPE_SECRET_EXPOSED',
35
+ 'X_SCRAPE_HTTP_FAILED',
36
+ 'X_SCRAPE_AUTH_FAILED',
37
+ 'X_SCRAPE_SESSION_EXPIRED',
38
+ 'X_SCRAPE_PROMPT_UNANSWERED',
39
+ 'X_SCRAPE_BLOCKED',
40
+ ] as const;
41
+
42
+ /**
43
+ * `X_NOT_IMPLEMENTED` is `@ultimat3/core`'s and `recover.ts` throws it for the agent seam, the
44
+ * way `packages/jobs/src/driver-redis.ts` does for its stub. `X_ENV_MISSING` is core's too: a
45
+ * declared secret with no value in the environment is a missing environment variable, not a
46
+ * scraping concept needing its own code. No title is kept here for either — one code, one owner,
47
+ * or the two copies drift.
48
+ */
49
+ export const SCRAPE_BORROWED_ERROR_CODES = ['X_NOT_IMPLEMENTED', 'X_ENV_MISSING'] as const;
50
+
51
+ export const SCRAPE_ERROR_CODES = [
52
+ ...SCRAPE_OWNED_ERROR_CODES,
53
+ ...SCRAPE_BORROWED_ERROR_CODES,
54
+ ] as const;
55
+
56
+ export type ScrapeOwnedErrorCode = (typeof SCRAPE_OWNED_ERROR_CODES)[number];
57
+ export type ScrapeErrorCode = (typeof SCRAPE_ERROR_CODES)[number];
58
+
59
+ export const SCRAPE_ERROR_TITLES: Readonly<Record<ScrapeOwnedErrorCode, string>> = {
60
+ X_SCRAPE_DRIVER_UNKNOWN: 'no browser driver is installed for this run',
61
+ X_SCRAPE_CDP_ATTACH_FAILED: 'the CDP endpoint refused the attach',
62
+ X_SCRAPE_BROWSER_UNREACHABLE: 'the browser went away mid-run',
63
+ X_SCRAPE_PROFILE_LOCKED: 'another process holds this browser profile',
64
+ X_SCRAPE_HOST_BLOCKED: 'the page asked for a host allowHosts does not list',
65
+ X_SCRAPE_SELECTOR_MISSING: 'the selector never appeared inside its window',
66
+ X_SCRAPE_NOT_ACTIONABLE: 'the element is present and cannot be acted on',
67
+ X_SCRAPE_TIMEOUT: 'the step exceeded its wall-clock budget',
68
+ X_SCRAPE_WEDGED: 'the browser stopped answering and was killed',
69
+ X_SCRAPE_PAGE_CRASHED: 'the renderer process died',
70
+ X_SCRAPE_OUTPUT_INVALID: 'the extracted rows do not match the extract schema',
71
+ X_SCRAPE_YIELD_COLLAPSED: 'the run succeeded and returned far too little',
72
+ X_SCRAPE_DOWNLOAD_TIMEOUT: 'the download never landed',
73
+ X_SCRAPE_ROBOTS_DISALLOWED: 'robots.txt disallows this path',
74
+ X_SCRAPE_FIXTURE_MISSING: 'the fixture driver has no recording for this request',
75
+ X_SCRAPE_FIXTURE_STALE: 'the recording is older than the fixture max age',
76
+ X_SCRAPE_REMOTE_REQUIRED: 'this driver needs a cdpUrl and was given none',
77
+ X_SCRAPE_RECOVER_REFUSED: 'the recovery hook declined to recover this failure',
78
+ X_SCRAPE_SECRET_EXPOSED: 'an artifact would have carried a secret this run typed',
79
+ X_SCRAPE_HTTP_FAILED: 'the site answered the HTTP leg with a non-2xx status',
80
+ X_SCRAPE_AUTH_FAILED: 'the credentials were rejected',
81
+ X_SCRAPE_SESSION_EXPIRED: 'the restored session is no longer valid and nothing can renew it',
82
+ X_SCRAPE_PROMPT_UNANSWERED: 'a login step asked for a code and nothing answered',
83
+ X_SCRAPE_BLOCKED: 'the site refused this client — the identity is spent',
84
+ };
85
+
86
+ // One unconditional call, so a second package claiming one of these codes throws
87
+ // X_ERROR_CODE_DUPLICATE instead of losing silently to whichever module imported first.
88
+ registerErrorCodes(
89
+ Object.fromEntries(Object.entries(SCRAPE_ERROR_TITLES).map(([code, title]) => [code, { title }])),
90
+ );
91
+
92
+ /**
93
+ * The retry table, which is this package's most load-bearing declaration: a worker reads it to
94
+ * decide whether attempt 2 is worth running, and both wrong answers cost real money.
95
+ *
96
+ * The rule used, stated once: **retryable means the same code, run again, has a real chance of a
97
+ * different answer.** Transport, budget and liveness faults qualify. Everything that says "the
98
+ * page is not the page this scraper was written against" does not — attempt 5 hammers a site with
99
+ * a request that cannot succeed, and on an authenticated target that is how three wrong attempts
100
+ * lock an account. `X_SCRAPE_YIELD_COLLAPSED` is deliberately terminal for the same reason: a run
101
+ * that returned nothing is a human's problem, not a queue's.
102
+ */
103
+ export const SCRAPE_ERROR_RETRY = {
104
+ X_SCRAPE_CDP_ATTACH_FAILED: 'retryable',
105
+ X_SCRAPE_BROWSER_UNREACHABLE: 'retryable',
106
+ X_SCRAPE_TIMEOUT: 'retryable',
107
+ X_SCRAPE_WEDGED: 'retryable',
108
+ X_SCRAPE_DOWNLOAD_TIMEOUT: 'retryable',
109
+ // Retryable AND it burns the session first (`scrape-run.ts`). Retrying a block on the SAME
110
+ // persisted identity re-trips it every time: the flagged cookies are the thing being refused,
111
+ // so the retry has to arrive as somebody else or it is arithmetic, not a retry.
112
+ X_SCRAPE_BLOCKED: 'retryable',
113
+ // Non-2xx is transient far more often than not (429, 502, a deploy). A 4xx that is genuinely
114
+ // permanent is thrown with a per-instance `terminal` override, which `UltimateError` supports —
115
+ // one code, and the throw site decides, because the same status is both at different sites.
116
+ X_SCRAPE_HTTP_FAILED: 'retryable',
117
+ // Everything below is terminal, and each one is listed rather than left to the default so that
118
+ // deleting a line is a visible decision.
119
+ X_SCRAPE_DRIVER_UNKNOWN: 'terminal',
120
+ X_SCRAPE_PROFILE_LOCKED: 'terminal',
121
+ X_SCRAPE_HOST_BLOCKED: 'terminal',
122
+ X_SCRAPE_SELECTOR_MISSING: 'terminal',
123
+ X_SCRAPE_NOT_ACTIONABLE: 'terminal',
124
+ // A renderer that died takes its tab's state with it. Every retry so far in this package is a
125
+ // retry of an attempt that could still be somewhere; this one cannot, and a re-run of a
126
+ // half-submitted form is the incident, not the recovery.
127
+ X_SCRAPE_PAGE_CRASHED: 'terminal',
128
+ X_SCRAPE_OUTPUT_INVALID: 'terminal',
129
+ X_SCRAPE_YIELD_COLLAPSED: 'terminal',
130
+ X_SCRAPE_ROBOTS_DISALLOWED: 'terminal',
131
+ X_SCRAPE_FIXTURE_MISSING: 'terminal',
132
+ X_SCRAPE_FIXTURE_STALE: 'terminal',
133
+ X_SCRAPE_REMOTE_REQUIRED: 'terminal',
134
+ X_SCRAPE_RECOVER_REFUSED: 'terminal',
135
+ X_SCRAPE_SECRET_EXPOSED: 'terminal',
136
+ // THE hard rule of this package. A site that locks an account after three wrong attempts turns
137
+ // a retrying framework into the thing that destroys the user's account — so a rejected
138
+ // credential is terminal, always, and no retry policy, recovery hook or backoff can reach it.
139
+ X_SCRAPE_AUTH_FAILED: 'terminal',
140
+ X_SCRAPE_SESSION_EXPIRED: 'terminal',
141
+ X_SCRAPE_PROMPT_UNANSWERED: 'terminal',
142
+ } as const satisfies Readonly<Record<ScrapeOwnedErrorCode, 'retryable' | 'terminal'>>;
143
+
144
+ registerErrorRetry(SCRAPE_ERROR_RETRY);
145
+
146
+ export interface ScrapeErrorInit {
147
+ readonly code: ScrapeErrorCode;
148
+ readonly cause: string;
149
+ readonly fix: string;
150
+ readonly meta?: Readonly<Record<string, unknown>> | undefined;
151
+ /**
152
+ * Per-instance override of the table above, for the one case where a code is genuinely both:
153
+ * a 429 and a 404 share `X_SCRAPE_HTTP_FAILED`, and only the throw site knows which it saw.
154
+ */
155
+ readonly retry?: ErrorRetry | undefined;
156
+ }
157
+
158
+ export class ScrapeError extends UltimateError {
159
+ override readonly name = 'ScrapeError';
160
+
161
+ constructor(init: ScrapeErrorInit) {
162
+ super({
163
+ code: init.code,
164
+ cause: init.cause,
165
+ fix: init.fix,
166
+ docs: errorDocsUrl(init.code),
167
+ meta: init.meta,
168
+ ...(init.retry === undefined ? {} : { retry: init.retry }),
169
+ });
170
+ }
171
+ }
172
+
173
+ export function isScrapeError(value: unknown): value is ScrapeError {
174
+ return value instanceof ScrapeError;
175
+ }
176
+
177
+ /** True when the failure is one a second attempt could survive — the table above, read back. */
178
+ export function isRetryableScrapeError(value: unknown): boolean {
179
+ return isScrapeError(value) && value.retry !== 'terminal';
180
+ }
package/src/events.ts ADDED
@@ -0,0 +1,74 @@
1
+ // The run's event stream — and it goes through `@ultimat3/core`'s `Logger`, which `Ctx` already
2
+ // carries. This package ships CALL SITES and a FIELD VOCABULARY; it ships no sink, no transport
3
+ // and no second logger. Where the lines go is the app's existing logger configuration (axiom 1:
4
+ // one way to do each thing, and logging already has one).
5
+ //
6
+ // The event stream is the observability model: a debug report, an operator asking "where did it
7
+ // stop", and any future recovery pass all read the same lines. So every step emits start, then
8
+ // exactly one of success or failure, with its duration.
9
+ //
10
+ // What may be in a field is a CLOSED type. That is the mechanism that keeps a session cookie out
11
+ // of a log line: `ScrapeEventFields` has no key to put one in, and core's logger redacts the
12
+ // remaining spellings by key and every `Secret` by value.
13
+
14
+ import type { Logger } from '@ultimat3/core';
15
+ import type { ScrapeClock } from './clock';
16
+ import { errorCode } from './failures';
17
+
18
+ export interface ScrapeEventFields {
19
+ readonly scrape?: string | undefined;
20
+ readonly runId?: string | undefined;
21
+ readonly attempt?: number | undefined;
22
+ readonly step?: string | undefined;
23
+ readonly durationMs?: number | undefined;
24
+ readonly rows?: number | undefined;
25
+ readonly driver?: string | undefined;
26
+ readonly url?: string | undefined;
27
+ readonly code?: string | undefined;
28
+ /** Shape of a session, never its content — `sessionDigest()` builds these. */
29
+ readonly session?: string | undefined;
30
+ readonly cookies?: number | undefined;
31
+ readonly storageKeys?: number | undefined;
32
+ readonly origin?: string | undefined;
33
+ readonly refused?: number | undefined;
34
+ readonly reused?: boolean | undefined;
35
+ readonly burned?: boolean | undefined;
36
+ }
37
+
38
+ /** One child logger per run, so every line downstream carries the run's identity unasked. */
39
+ export const scrapeLogger = (logger: Logger, fields: ScrapeEventFields): Logger =>
40
+ logger.child({ ...fields, component: 'scrape' });
41
+
42
+ export interface StepEvent {
43
+ readonly name: string;
44
+ readonly logger: Logger;
45
+ readonly clock: ScrapeClock;
46
+ readonly attempt?: number | undefined;
47
+ }
48
+
49
+ /**
50
+ * Start, then exactly one of success or failure, with a duration. The failure line carries the
51
+ * error's CODE and never its message: a code is stable, greppable and safe, and a message is
52
+ * whatever a site put in its HTML.
53
+ */
54
+ export async function withStepEvent<T>(event: StepEvent, run: () => Promise<T>): Promise<T> {
55
+ const startedAt = event.clock.monotonic();
56
+ event.logger.debug('scrape.step.start', { step: event.name, attempt: event.attempt });
57
+ try {
58
+ const result = await run();
59
+ event.logger.info('scrape.step.ok', {
60
+ step: event.name,
61
+ attempt: event.attempt,
62
+ durationMs: Math.round(event.clock.monotonic() - startedAt),
63
+ });
64
+ return result;
65
+ } catch (thrown) {
66
+ event.logger.warn('scrape.step.failed', {
67
+ step: event.name,
68
+ attempt: event.attempt,
69
+ durationMs: Math.round(event.clock.monotonic() - startedAt),
70
+ code: errorCode(thrown),
71
+ });
72
+ throw thrown;
73
+ }
74
+ }
package/src/expect.ts ADDED
@@ -0,0 +1,133 @@
1
+ // The silent-green alarm. The worst failure a scraper has is not a crash — it is a run that
2
+ // succeeds, returns nothing, writes nothing, alerts nobody, and stays green until somebody asks
3
+ // where the data went. `expect` turns that into a red run: a yield under `minRows`, or under
4
+ // `maxDrop` of what this scrape normally returns, throws.
5
+
6
+ import { yieldCollapsed } from './error-throws';
7
+ import type { ScrapeError } from './errors';
8
+
9
+ export interface YieldExpectation {
10
+ /**
11
+ * The floor, in rows. A scrape whose real answer is legitimately sometimes zero declares
12
+ * `minRows: 0` — explicitly, so the reader can tell "zero is fine here" from "nobody thought
13
+ * about it". Omitted means only `maxDrop` applies.
14
+ */
15
+ readonly minRows?: number | undefined;
16
+ /**
17
+ * How far below the trailing median a run may fall, as a fraction: `0.5` allows half. The
18
+ * comparison is against the MEDIAN and not the mean because one 12,000-row backfill run in the
19
+ * history would drag a mean high enough to fire the alarm on every ordinary day afterwards.
20
+ */
21
+ readonly maxDrop?: number | undefined;
22
+ /** How many past runs form the baseline. */
23
+ readonly window?: number | undefined;
24
+ }
25
+
26
+ export const DEFAULT_YIELD_WINDOW = 7;
27
+
28
+ /**
29
+ * Below this many samples there is no baseline, so `maxDrop` cannot fire. A median of one run
30
+ * would make the SECOND run of a brand-new scraper alarm on any variation at all, and an alarm
31
+ * that fires on day two of every new scraper is an alarm somebody turns off.
32
+ */
33
+ export const MIN_BASELINE_RUNS = 3;
34
+
35
+ /** What a run's yield is measured against. Persisted by the app, never by this package. */
36
+ export interface YieldHistory {
37
+ /** Most recent first or last — order is irrelevant to a median, and this says so. */
38
+ recent(scrape: string, limit: number): Promise<readonly number[]>;
39
+ record(scrape: string, rows: number): Promise<void>;
40
+ }
41
+
42
+ export function median(values: readonly number[]): number | undefined {
43
+ if (values.length === 0) return undefined;
44
+ const sorted = [...values].sort((a, b) => a - b);
45
+ const mid = Math.floor(sorted.length / 2);
46
+ const low = sorted[mid - 1];
47
+ const high = sorted[mid];
48
+ if (high === undefined) return undefined;
49
+ return sorted.length % 2 === 1 ? high : ((low ?? high) + high) / 2;
50
+ }
51
+
52
+ export interface YieldCheck {
53
+ readonly scrape: string;
54
+ readonly rows: number;
55
+ readonly expect: YieldExpectation;
56
+ readonly history: readonly number[];
57
+ }
58
+
59
+ /**
60
+ * The whole rule, pure: the error this yield earns, or `undefined`. Pure so a test can hand it
61
+ * fifty histories without a queue, a browser or a clock.
62
+ */
63
+ export function yieldProblem(check: YieldCheck): ScrapeError | undefined {
64
+ const { minRows, maxDrop } = check.expect;
65
+ if (minRows !== undefined && check.rows < minRows) {
66
+ return yieldCollapsed({ scrape: check.scrape, rows: check.rows, reason: 'min-rows', minRows });
67
+ }
68
+ if (maxDrop === undefined || check.history.length < MIN_BASELINE_RUNS) return undefined;
69
+ const baseline = median(check.history);
70
+ if (baseline === undefined || baseline <= 0) return undefined;
71
+ if (check.rows >= baseline * (1 - maxDrop)) return undefined;
72
+ return yieldCollapsed({
73
+ scrape: check.scrape,
74
+ rows: check.rows,
75
+ reason: 'drop',
76
+ baseline,
77
+ maxDrop,
78
+ });
79
+ }
80
+
81
+ export interface YieldGuardInput {
82
+ readonly scrape: string;
83
+ readonly rows: number;
84
+ readonly expect: YieldExpectation | undefined;
85
+ readonly history: YieldHistory | undefined;
86
+ }
87
+
88
+ /**
89
+ * Check, then record — and record ONLY a run that passed.
90
+ *
91
+ * That order is the mechanism, not an implementation detail. Recording a collapsed run would let
92
+ * the baseline follow the collapse down: three broken runs at 2 rows and the median IS 2, so the
93
+ * fourth broken run is within `maxDrop` of it and the alarm has silenced itself exactly when it
94
+ * was working. A scraper that quietly re-baselines onto its own failure is the bug this file
95
+ * exists to prevent, one level up.
96
+ */
97
+ export async function guardYield(input: YieldGuardInput): Promise<void> {
98
+ // No `expect` is no baseline either, deliberately: with no floor and no drop rule there is
99
+ // nothing deciding whether a run was good, so recording it would let a stretch of silent
100
+ // zero-row runs become the median an `expect` added later is measured against. The cost is that
101
+ // `maxDrop` needs `MIN_BASELINE_RUNS` runs after it is declared before it can fire — a delay,
102
+ // not a hole. `expect.test.ts` pins both halves.
103
+ if (input.expect === undefined) return;
104
+ const window = input.expect.window ?? DEFAULT_YIELD_WINDOW;
105
+ const history =
106
+ input.history === undefined ? [] : await input.history.recent(input.scrape, window);
107
+ const problem = yieldProblem({
108
+ scrape: input.scrape,
109
+ rows: input.rows,
110
+ expect: input.expect,
111
+ history,
112
+ });
113
+ if (problem !== undefined) throw problem;
114
+ await input.history?.record(input.scrape, input.rows);
115
+ }
116
+
117
+ /** In-memory history: what `fakeBrowser()` runs against, and what a test asserts on. */
118
+ export function memoryYieldHistory(
119
+ seed: Readonly<Record<string, readonly number[]>> = {},
120
+ ): YieldHistory {
121
+ const runs = new Map<string, number[]>(
122
+ Object.entries(seed).map(([name, values]) => [name, [...values]]),
123
+ );
124
+ return {
125
+ recent: (scrape, limit) => Promise.resolve((runs.get(scrape) ?? []).slice(-limit)),
126
+ record: (scrape, rows) => {
127
+ const existing = runs.get(scrape) ?? [];
128
+ existing.push(rows);
129
+ runs.set(scrape, existing);
130
+ return Promise.resolve();
131
+ },
132
+ };
133
+ }