@ultimat3/scraping 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 developerz.ai
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md ADDED
@@ -0,0 +1,194 @@
1
+ # @ultimat3/scraping
2
+
3
+ Browser automation as a **job**. `scrape()` returns a `JobHandle` — the rule's fourth instance
4
+ after `llm()` (an action factory) and `backfill()` (a job factory). There is no ninth primitive.
5
+
6
+ ```ts
7
+ import { t } from '@ultimat3/schema';
8
+ import type { StorageDriver } from '@ultimat3/storage';
9
+ import { scrape, storageSessionStore } from '@ultimat3/scraping';
10
+
11
+ declare const disk: StorageDriver;
12
+
13
+ const orderPage = t.object({
14
+ rows: t.array(t.object({ id: t.string, total: t.number })),
15
+ hasMore: t.boolean,
16
+ });
17
+
18
+ export const dailyOrders = scrape({
19
+ name: 'orders.daily',
20
+ input: t.object({ orgId: t.uuid, day: t.string }),
21
+ extract: t.object({ id: t.string, total: t.number }),
22
+ idempotencyKey: ({ orgId, day }) => `orders:${orgId}:${day}`,
23
+ tenant: ({ orgId }) => orgId,
24
+ allowHosts: ['shop.example.com', '*.api.example.com'],
25
+ block: ['image', 'media', 'font'],
26
+ rate: 1,
27
+ robots: 'obey',
28
+ secrets: ['SHOP_PASSWORD'],
29
+ expect: { minRows: 1, maxDrop: 0.5 },
30
+ auth: {
31
+ store: storageSessionStore(disk),
32
+ login: async ({ page, secrets, prompt }) => {
33
+ await page.goto('https://shop.example.com/login');
34
+ await page.fill('#user', 'ops@example.com');
35
+ await page.fill('#pass', secrets.get('SHOP_PASSWORD'));
36
+ await page.click('#submit');
37
+ await page.fill('#otp', await prompt('sms code'));
38
+ },
39
+ validate: async ({ page }) => (await page.count('#logout')) > 0,
40
+ },
41
+ async run({ page, http, step }) {
42
+ await page.goto('https://shop.example.com/orders');
43
+ const rows: { id: string; total: number }[] = [];
44
+ for (let pageNumber = 1; ; pageNumber += 1) {
45
+ // One `step.run` per page: a killed worker resumes at the page it stopped on. What a step
46
+ // persists is a CURSOR — never a page, never a live handle, never a session.
47
+ const batch = await step.run(`page:${pageNumber}`, async () =>
48
+ (await http.request(`https://api.example.com/orders?page=${pageNumber}`)).parse(orderPage),
49
+ );
50
+ rows.push(...batch.rows);
51
+ if (!batch.hasMore) break;
52
+ }
53
+ return rows;
54
+ },
55
+ });
56
+ ```
57
+
58
+ ## Why a job
59
+
60
+ | A scrape has | So it is |
61
+ |---|---|
62
+ | an input schema, a tenant, a retry policy, a timeout, a queue, a concurrency cap | a `job` |
63
+ | a **required** idempotency key | a `job` — re-logging into a bank after a worker kill is the exact bug; three wrong attempts locks the account |
64
+ | `step.run` checkpoints, because recovery must resume at the broken page | a `job` |
65
+
66
+ It therefore inherits `.enqueue()`, the worker's cancellation, the dead-letter path, `x jobs show`
67
+ and its manifest row without a line of code here.
68
+
69
+ ## The two transports, one session
70
+
71
+ Drive the **browser** through login, 2FA and navigation; then reverse-engineer the site's own JSON
72
+ endpoints and pull the bulk over **HTTP**. `http` is session-bound, never a bare `fetch`: the
73
+ browser's cookies, headers and proxy, the same `allowHosts`, the same robots gate, the same rate
74
+ limit, the same cancellation. `page.session()` exposes the handoff so an author can see what
75
+ carried over.
76
+
77
+ The jar is scoped per request, RFC 6265: a cookie stored for `bank.test` is sent to `bank.test`
78
+ and to nothing else — not `evilbank.test`, not `sub.bank.test` — and only a domain-scoped
79
+ `.bank.test` reaches subdomains. `SessionSnapshot.headers` is the one field the real driver cannot
80
+ fill (CDP exposes no read for it, and `driver-parity.test.ts` pins that): a token the HTTP leg must
81
+ carry goes on the request, `http.request(url, { headers })`.
82
+
83
+ Both legs replay from **one** fixture directory (`fixtureBrowser(dir)`), so a hybrid run — browser
84
+ login, session handoff, HTTP bulk fetch — is tested end to end. Both legs apply the same robots
85
+ gate, offline included.
86
+
87
+ ## What it owns
88
+
89
+ | Module | Owns |
90
+ |---|---|
91
+ | `scrape.ts` / `scrape-run.ts` | the factory over `job()`, and one attempt's assembly |
92
+ | `page.ts` / `page-over-target.ts` | the driver-blind vocabulary, implemented once |
93
+ | `target.ts` / `driver.ts` | the two seams: what a driver answers, and what a session is |
94
+ | `driver-cdp.ts` / `cdp-*.ts` | the real browser, over a structural CDP port |
95
+ | `driver-fake.ts` / `driver-fixture.ts` / `html-*.ts` | the offline drivers, on Bun's `HTMLRewriter` |
96
+ | `http.ts` / `http-recorded.ts` | the second transport, live and replayed |
97
+ | `auth.ts` / `session-state.ts` | acquire → persist → reuse → validate → burn |
98
+ | `expect.ts` | the silent-green alarm |
99
+ | `watchdog.ts` | the wedge and zombie discipline |
100
+ | `errors.ts` / `error-throws.ts` | this package's `X_*` codes and their retry classification |
101
+
102
+ ## Extending it — there is no plugin API, and none is needed
103
+
104
+ Two mechanisms, both of which the framework already ships:
105
+
106
+ **1. The driver seam.** `ScrapeDriver` is the extension point, exactly as `jobs`' driver is.
107
+ Everything a third-party driver needs is a named export of `src/index.ts`:
108
+
109
+ ```ts
110
+ import {
111
+ pageOverTarget,
112
+ type ScrapeDriver,
113
+ type ScrapeHttp,
114
+ type ScrapeSession,
115
+ type ScrapeTarget,
116
+ type SessionInit,
117
+ } from '@ultimat3/scraping';
118
+
119
+ interface MyOptions {
120
+ readonly endpoint: string;
121
+ }
122
+
123
+ // Implement `ScrapeTarget` — twelve methods, all driver-blind — and the vocabulary above it,
124
+ // actionability and frame re-resolution included, comes from `pageOverTarget` unchanged.
125
+ declare function myTarget(options: MyOptions, init: SessionInit): Promise<ScrapeTarget>;
126
+ declare function myHttp(target: ScrapeTarget, init: SessionInit): ScrapeHttp;
127
+
128
+ export const myBrowser = (options: MyOptions): ScrapeDriver => ({
129
+ name: 'my-browser',
130
+ async open(init: SessionInit): Promise<ScrapeSession> {
131
+ const target = await myTarget(options, init);
132
+ return {
133
+ driver: 'my-browser',
134
+ page: pageOverTarget(target, {
135
+ clock: init.clock,
136
+ allowHosts: init.rules.allowHosts,
137
+ defaultTimeoutMs: init.timeoutMs,
138
+ secrets: init.secrets,
139
+ robots: init.robots,
140
+ signal: init.signal,
141
+ }),
142
+ http: myHttp(target, init),
143
+ close: () => target.close(),
144
+ };
145
+ },
146
+ });
147
+ ```
148
+
149
+ **2. Primitives are functions returning values.** An app's house style is a wrapper, not a fork:
150
+
151
+ ```ts
152
+ // apps/web/shared/base/bank-scrape.ts — the app's convention, written once
153
+ import type { JobHandle } from '@ultimat3/jobs';
154
+ import type { StorageDriver } from '@ultimat3/storage';
155
+ import { scrape, type ScrapeDefinition } from '@ultimat3/scraping';
156
+
157
+ declare const disk: StorageDriver;
158
+
159
+ export const bankScrape = <I, R>(over: ScrapeDefinition<I, R>): JobHandle<I> =>
160
+ scrape<I, R>({
161
+ robots: 'obey',
162
+ rate: 0.5,
163
+ block: ['image', 'media', 'font'],
164
+ expect: { minRows: 1, maxDrop: 0.5 },
165
+ artifacts: { storage: disk },
166
+ ...over,
167
+ });
168
+ ```
169
+
170
+ Nothing downstream can tell the difference: the value is still a `JobHandle`, so the registry, the
171
+ manifest and `x jobs show` work on it unchanged.
172
+
173
+ Where a genuine hook is needed it is a **declared callback field** on the definition — `recover`,
174
+ `auth.login`, `auth.validate`, `prompt` — typed, discoverable, one per concern. Never a global
175
+ register-a-plugin call ([`docs/idea/19-mechanism-not-convention.md`](../../docs/idea/19-mechanism-not-convention.md)).
176
+
177
+ ## What does NOT ship, and why
178
+
179
+ | Not shipped | Reason |
180
+ |---|---|
181
+ | stealth payloads | a framework-shipped payload is a shared, fingerprintable signature handed to every user. Two independent teams reached the same conclusion — one moved stealth into a Chromium fork, another *removed* an injected script because the injection was itself detectable. The hook ships; the payload never does |
182
+ | captcha solving | same argument, plus it is not a mechanism — it is a business decision about somebody else's site |
183
+ | credential stuffing, lockout-defeating retry | `X_SCRAPE_AUTH_FAILED` is **terminal**, and the refusal is written into the session record so the next attempt fails before reaching a login form. A site that locks an account after three wrong attempts makes a retrying framework the thing that destroys the user's account |
184
+ | a `puppeteer-core` dependency | the launcher is passed in (`localBrowser({ launcher: puppeteer })`), the port is structural, and no puppeteer type can reach the vocabulary |
185
+
186
+ ## Errors
187
+
188
+ Every code carries a cause, a runnable `fix:` and a retry classification — see `src/errors.ts`.
189
+ `X_SCRAPE_YIELD_COLLAPSED`, `X_SCRAPE_AUTH_FAILED`, `X_SCRAPE_BLOCKED` and `X_SCRAPE_PAGE_CRASHED`
190
+ are the four worth knowing by heart.
191
+
192
+ ## Boundary
193
+
194
+ Tier 5. May import tiers 0-4 only — enforced by `bun run scripts/boundaries.ts`.
package/package.json ADDED
@@ -0,0 +1,38 @@
1
+ {
2
+ "name": "@ultimat3/scraping",
3
+ "version": "2.0.0",
4
+ "description": "Browser automation as a job: scrape() returns a JobHandle",
5
+ "license": "MIT",
6
+ "type": "module",
7
+ "repository": {
8
+ "type": "git",
9
+ "url": "git+https://github.com/developerz-ai/ultimate.git",
10
+ "directory": "packages/scraping"
11
+ },
12
+ "publishConfig": {
13
+ "access": "public",
14
+ "provenance": true
15
+ },
16
+ "exports": {
17
+ ".": "./src/index.ts"
18
+ },
19
+ "files": [
20
+ "src",
21
+ "!src/**/*.test.ts",
22
+ "README.md",
23
+ "LICENSE"
24
+ ],
25
+ "engines": {
26
+ "bun": ">=1.3.0"
27
+ },
28
+ "scripts": {
29
+ "typecheck": "tsc --noEmit -p tsconfig.json",
30
+ "test": "bun test"
31
+ },
32
+ "dependencies": {
33
+ "@ultimat3/core": "2.0.0",
34
+ "@ultimat3/jobs": "2.0.0",
35
+ "@ultimat3/schema": "2.0.0",
36
+ "@ultimat3/storage": "2.0.0"
37
+ }
38
+ }
@@ -0,0 +1,106 @@
1
+ // What "the element is ready" means, in one place. `page.click` in a raw driver waits for the
2
+ // SELECTOR and nothing else, so every app that has ever used one re-learns the same lesson and
3
+ // writes the same `waitForTimeout(800)` — or the same hand-rolled
4
+ // `waitForSelector('#aceptar:not([disabled])')`, which is this rule, spelled once, by hand, at
5
+ // one call site out of forty.
6
+
7
+ import type { ScrapeClock } from './clock';
8
+ import { deadline } from './clock';
9
+ import { notActionable, selectorMissing } from './error-throws';
10
+ import type { ElementSnapshot } from './target';
11
+
12
+ /** What a caller needs to be true before it acts. Each level implies the ones above it. */
13
+ export type ActionabilityState = 'attached' | 'visible' | 'enabled' | 'actionable';
14
+
15
+ export const DEFAULT_POLL_MS = 50;
16
+
17
+ const sameBox = (a: ElementSnapshot, b: ElementSnapshot): boolean =>
18
+ a.box === undefined || b.box === undefined
19
+ ? true
20
+ : a.box.x === b.box.x &&
21
+ a.box.y === b.box.y &&
22
+ a.box.width === b.box.width &&
23
+ a.box.height === b.box.height;
24
+
25
+ /**
26
+ * Stability, as an observation rather than as a timer: two consecutive polls that agree on
27
+ * everything this package can see. A layout-carrying driver compares boxes and so catches an
28
+ * element still sliding in from a CSS transition; a DOM-only driver compares text, value and
29
+ * attributes and so catches one still being re-rendered. Neither sleeps 800ms and hopes.
30
+ */
31
+ export const isStable = (
32
+ current: ElementSnapshot,
33
+ previous: ElementSnapshot | undefined,
34
+ ): boolean =>
35
+ previous !== undefined &&
36
+ sameBox(current, previous) &&
37
+ current.text === previous.text &&
38
+ current.value === previous.value &&
39
+ current.enabled === previous.enabled &&
40
+ current.visible === previous.visible;
41
+
42
+ /**
43
+ * Why this snapshot is not yet ready for `state`, or `undefined` when it is.
44
+ *
45
+ * `hitTarget === undefined` means the driver has no layout engine, so nothing here can decide
46
+ * whether something covers the element — and it is NOT read as "covered". The divergence is
47
+ * deliberate, and `driver-parity.test.ts` pins it in one place: a fake that answered `false` would
48
+ * make every click in every offline test fail, and one that fabricated `true` on a real browser
49
+ * would hide the cookie banner that eats the click.
50
+ */
51
+ export function actionabilityProblem(
52
+ current: ElementSnapshot,
53
+ previous: ElementSnapshot | undefined,
54
+ state: ActionabilityState,
55
+ ): string | undefined {
56
+ if (state === 'attached') return undefined;
57
+ if (!current.visible) return 'not visible';
58
+ if (current.box !== undefined && (current.box.width === 0 || current.box.height === 0))
59
+ return 'has a zero-sized box';
60
+ if (state === 'visible') return undefined;
61
+ if (!current.enabled) return 'disabled';
62
+ if (state === 'enabled') return undefined;
63
+ if (current.hitTarget === false) return 'covered by another element at its centre';
64
+ if (!isStable(current, previous)) return 'still moving';
65
+ return undefined;
66
+ }
67
+
68
+ export interface ActionabilityWait {
69
+ readonly selector: string;
70
+ readonly url: string;
71
+ readonly state: ActionabilityState;
72
+ readonly timeoutMs: number;
73
+ readonly clock: ScrapeClock;
74
+ readonly pollMs?: number | undefined;
75
+ readonly signal?: AbortSignal | undefined;
76
+ /** Re-read from the live page on EVERY poll. Never a handle captured before the loop. */
77
+ snapshot(): Promise<ElementSnapshot | undefined>;
78
+ }
79
+
80
+ /**
81
+ * Poll until the element reaches `state`, or refuse with the reason it never did. Two codes and
82
+ * they are genuinely different questions: `X_SCRAPE_SELECTOR_MISSING` says the page does not have
83
+ * this element (the markup changed), `X_SCRAPE_NOT_ACTIONABLE` says it does and something is in
84
+ * the way (a modal, a spinner, a disabled submit). Collapsing them into one timeout is what makes
85
+ * a scraper failure take an afternoon.
86
+ */
87
+ export async function awaitActionable(wait: ActionabilityWait): Promise<ElementSnapshot> {
88
+ const budget = deadline(wait.clock, wait.timeoutMs);
89
+ const pollMs = wait.pollMs ?? DEFAULT_POLL_MS;
90
+ let previous: ElementSnapshot | undefined;
91
+ let lastProblem: string | undefined;
92
+ let everSeen = false;
93
+ for (;;) {
94
+ const current = await wait.snapshot();
95
+ if (current !== undefined) {
96
+ everSeen = true;
97
+ lastProblem = actionabilityProblem(current, previous, wait.state);
98
+ if (lastProblem === undefined) return current;
99
+ previous = current;
100
+ }
101
+ if (budget.expired()) break;
102
+ await wait.clock.sleep(Math.min(pollMs, budget.remainingMs()), wait.signal);
103
+ }
104
+ if (!everSeen) throw selectorMissing(wait.selector, wait.url, wait.timeoutMs);
105
+ throw notActionable(wait.selector, lastProblem ?? 'never became actionable', wait.timeoutMs);
106
+ }
@@ -0,0 +1,69 @@
1
+ // Run artifacts: the page HTML, the screenshot, the downloaded file. A scrape that failed and
2
+ // kept nothing is a scrape somebody has to reproduce by hand, and the page it failed on no longer
3
+ // exists by the time they look.
4
+ //
5
+ // Written through `@ultimat3/storage`'s driver, so the same call lands on a local disk in
6
+ // development and on S3 in production, and this package owns no upload path of its own.
7
+
8
+ import type { StorageDriver } from '@ultimat3/storage';
9
+
10
+ export interface ArtifactRef {
11
+ readonly key: string;
12
+ readonly bytes: number;
13
+ readonly contentType: string;
14
+ }
15
+
16
+ export interface ArtifactWriter {
17
+ /** Bytes under this run's own prefix. Answers the key, which is what a report links to. */
18
+ save(name: string, body: Uint8Array | string, contentType?: string): Promise<ArtifactRef>;
19
+ /** Everything written by this run, in order. Bounded by how many the body asked for. */
20
+ readonly saved: readonly ArtifactRef[];
21
+ }
22
+
23
+ export interface ArtifactWriterInit {
24
+ readonly storage: StorageDriver | undefined;
25
+ readonly scrape: string;
26
+ readonly runId: string;
27
+ /** Optional prefix, so an app can keep scrape artifacts out of its user-upload namespace. */
28
+ readonly prefix?: string | undefined;
29
+ }
30
+
31
+ export const DEFAULT_ARTIFACT_PREFIX = 'scrape';
32
+
33
+ const CONTENT_TYPES: Readonly<Record<string, string>> = {
34
+ html: 'text/html; charset=utf-8',
35
+ png: 'image/png',
36
+ pdf: 'application/pdf',
37
+ json: 'application/json',
38
+ csv: 'text/csv',
39
+ txt: 'text/plain; charset=utf-8',
40
+ };
41
+
42
+ export const contentTypeFor = (name: string): string =>
43
+ CONTENT_TYPES[name.split('.').pop()?.toLowerCase() ?? ''] ?? 'application/octet-stream';
44
+
45
+ /**
46
+ * A writer with no storage driver is a NO-OP that still answers a key — never a throw. An app
47
+ * that has not configured storage should still be able to run a scrape; losing the artifact is a
48
+ * cost, and refusing the run over it is a bigger one.
49
+ */
50
+ export function createArtifactWriter(init: ArtifactWriterInit): ArtifactWriter {
51
+ const saved: ArtifactRef[] = [];
52
+ const prefix = `${init.prefix ?? DEFAULT_ARTIFACT_PREFIX}/${init.scrape}/${init.runId}`;
53
+ return {
54
+ saved,
55
+ async save(name, body, contentType): Promise<ArtifactRef> {
56
+ const bytes = typeof body === 'string' ? new TextEncoder().encode(body) : body;
57
+ const ref: ArtifactRef = {
58
+ key: `${prefix}/${name}`,
59
+ bytes: bytes.byteLength,
60
+ contentType: contentType ?? contentTypeFor(name),
61
+ };
62
+ if (init.storage !== undefined) {
63
+ await init.storage.put(ref.key, bytes, { contentType: ref.contentType });
64
+ }
65
+ saved.push(ref);
66
+ return ref;
67
+ },
68
+ };
69
+ }
package/src/auth.ts ADDED
@@ -0,0 +1,200 @@
1
+ // Session lifecycle: acquire -> persist -> reuse -> validate -> burn. Authenticated scraping is
2
+ // the primary case, not an edge case, so this is a declared part of the API rather than something
3
+ // every app hand-rolls.
4
+ //
5
+ // The order below is the whole design, and each step is there because skipping it costs something
6
+ // real:
7
+ //
8
+ // | Step | Why it is not optional |
9
+ // |---|---|
10
+ // | reuse | logging in every run is slow, and the login itself is the anti-bot signal |
11
+ // | validate | an expired session must be caught by a cheap probe, not by a failure 40 steps later |
12
+ // | login | only when the probe says the session is gone — the author drives the browser |
13
+ // | persist | a 2FA code the user typed is worth keeping; that is the whole economic argument |
14
+ // | burn | a flagged identity stays flagged, so a retry on it re-trips the same block forever |
15
+ //
16
+ // What is NOT here, deliberately: credential stuffing, captcha solving, and any retry of a
17
+ // rejected credential. A rejected credential is `X_SCRAPE_AUTH_FAILED`, terminal, and the refusal
18
+ // is written into the session record so the NEXT attempt fails before reaching the login form.
19
+
20
+ import type { Logger } from '@ultimat3/core';
21
+ import type { ScrapeClock } from './clock';
22
+ import { authFailed, promptUnanswered, sessionExpired } from './error-throws';
23
+ import type { ScrapePage } from './page';
24
+ import type { ScrapeSecrets } from './secrets';
25
+ import { type ScrapeSessionStore, type SessionState, sessionDigest } from './session-state';
26
+
27
+ export interface PromptRequest {
28
+ /** What the site is asking for: `'sms code'`, `'authenticator code'`, `'security question 3'`. */
29
+ readonly label: string;
30
+ readonly scrape: string;
31
+ readonly url: string;
32
+ }
33
+
34
+ /**
35
+ * Where an out-of-band code comes from. A declared callback field on the definition — typed,
36
+ * discoverable, one per concern — and never a global "register a plugin" call.
37
+ */
38
+ export type PromptHandler = (request: PromptRequest) => Promise<string> | string;
39
+
40
+ export interface AuthContext<I> {
41
+ readonly input: I;
42
+ readonly page: ScrapePage;
43
+ /**
44
+ * The declared secrets, boxed. A login body needs the credential it is about to type, and
45
+ * without this field there was no way to reach one — the README's own first example could not
46
+ * have compiled, which is how the omission was found.
47
+ */
48
+ readonly secrets: ScrapeSecrets;
49
+ /** The out-of-band code seam. Throws `X_SCRAPE_PROMPT_UNANSWERED` if nothing was declared. */
50
+ prompt(label: string): Promise<string>;
51
+ }
52
+
53
+ export interface ScrapeAuth<I> {
54
+ /**
55
+ * Drives the browser through the login. Runs ONLY when there is no valid session — never on
56
+ * every run, and never as a retry of a refused credential.
57
+ */
58
+ login(context: AuthContext<I>): Promise<void>;
59
+ /**
60
+ * The cheap probe: is the restored session still good? Site knowledge, so the author declares
61
+ * it and the framework decides what to do with the answer. Omitted means a restored session is
62
+ * trusted until something else fails.
63
+ */
64
+ validate?(context: AuthContext<I>): Promise<boolean>;
65
+ /** Where sessions live. Omitted means no reuse at all — every run logs in. */
66
+ readonly store?: ScrapeSessionStore | undefined;
67
+ /**
68
+ * The DISCRIMINATOR inside this tenant's key space — a second account, say. It is one segment of
69
+ * `sessionKeyFor({ scrape, tenant, discriminator })` and never the whole key: a value that
70
+ * replaced the key would put two tenants declaring the same account name on one authenticated
71
+ * session. Sanitised like every other segment, so it cannot escape the key space either.
72
+ */
73
+ key?(input: I): string;
74
+ /** Reuse a stored session. `false` forces a fresh login every run. Defaults to `true`. */
75
+ readonly reuse?: boolean | undefined;
76
+ /** Milliseconds. A stored session older than this is not restored. */
77
+ readonly maxAge?: number | undefined;
78
+ }
79
+
80
+ export function createPrompt(
81
+ scrape: string,
82
+ handler: PromptHandler | undefined,
83
+ page: ScrapePage,
84
+ ): (label: string) => Promise<string> {
85
+ return async (label: string): Promise<string> => {
86
+ if (handler === undefined) throw promptUnanswered(scrape, label);
87
+ const answer = await handler({ label, scrape, url: page.url() });
88
+ if (typeof answer !== 'string' || answer === '') throw promptUnanswered(scrape, label);
89
+ return answer;
90
+ };
91
+ }
92
+
93
+ export interface AuthPlanInput<I> {
94
+ readonly scrape: string;
95
+ readonly auth: ScrapeAuth<I> | undefined;
96
+ readonly key: string;
97
+ readonly clock: ScrapeClock;
98
+ readonly logger: Logger;
99
+ }
100
+
101
+ /**
102
+ * The stored session this run may restore, or `undefined`.
103
+ *
104
+ * THROWS when the record carries a refusal: those credentials were rejected, and reaching the
105
+ * login form again with them is how an account gets locked. It runs BEFORE `driver.open()`, which
106
+ * is what it buys: `X_SCRAPE_AUTH_FAILED` is registered `terminal` so the queue will not retry,
107
+ * and this makes a replay, a manual requeue or a second enqueue refuse without spending a browser,
108
+ * a CDP attach or a request on an answer that cannot change.
109
+ */
110
+ export async function restorableSession<I>(
111
+ plan: AuthPlanInput<I>,
112
+ ): Promise<SessionState | undefined> {
113
+ const store = plan.auth?.store;
114
+ if (store === undefined) return undefined;
115
+ const found = await store.load(plan.key);
116
+ if (found === undefined) return undefined;
117
+ // The tombstone is read BEFORE `reuse` is honoured. `reuse: false` says "do not restore this
118
+ // session"; it does not say "present the rejected credential again", and reading it first meant
119
+ // a `reuse: false` scrape walked a refused password back to the login form on every requeue.
120
+ if (found.refusedAt !== undefined) throw authFailed(plan.scrape, `refused at ${found.refusedAt}`);
121
+ if (plan.auth?.reuse === false) return undefined;
122
+ const maxAge = plan.auth?.maxAge;
123
+ if (maxAge !== undefined) {
124
+ const age = plan.clock.now().getTime() - new Date(found.savedAt).getTime();
125
+ if (age > maxAge) return undefined;
126
+ }
127
+ plan.logger.debug('scrape.session.restored', sessionDigest(found));
128
+ return found;
129
+ }
130
+
131
+ /** Written down so the next attempt cannot reach the login form with the same credentials. */
132
+ export async function markRefused<I>(plan: AuthPlanInput<I>): Promise<void> {
133
+ const store = plan.auth?.store;
134
+ if (store === undefined) return;
135
+ await store.save({
136
+ key: plan.key,
137
+ savedAt: plan.clock.now().toISOString(),
138
+ refusedAt: plan.clock.now().toISOString(),
139
+ cookies: [],
140
+ headers: {},
141
+ storage: {},
142
+ userAgent: '',
143
+ origin: '',
144
+ });
145
+ }
146
+
147
+ /** New identity, from scratch. One call, because a flagged profile is unusable and must go. */
148
+ export async function burnSession<I>(plan: AuthPlanInput<I>): Promise<void> {
149
+ if (plan.auth?.store === undefined) return;
150
+ await plan.auth.store.burn(plan.key);
151
+ plan.logger.warn('scrape.session.burned', { burned: true });
152
+ }
153
+
154
+ export interface EnsureAuthInput<I> extends AuthPlanInput<I> {
155
+ readonly input: I;
156
+ readonly page: ScrapePage;
157
+ readonly secrets: ScrapeSecrets;
158
+ readonly restored: SessionState | undefined;
159
+ readonly prompt: (label: string) => Promise<string>;
160
+ }
161
+
162
+ /**
163
+ * Log in if — and only if — this run has no session it can prove is still good.
164
+ *
165
+ * A resumed attempt takes the same path and reaches the same conclusion: the session lives in the
166
+ * store, not in a step record, so restarting the job re-reads it and re-probes it rather than
167
+ * replaying a checkpoint that says "logged in" about a session that has since expired. What a
168
+ * step may persist is a cursor; a session is neither a cursor nor a page.
169
+ */
170
+ export async function ensureAuthenticated<I>(args: EnsureAuthInput<I>): Promise<boolean> {
171
+ const auth = args.auth;
172
+ if (auth === undefined) return false;
173
+ const context: AuthContext<I> = {
174
+ input: args.input,
175
+ page: args.page,
176
+ secrets: args.secrets,
177
+ prompt: args.prompt,
178
+ };
179
+ if (args.restored !== undefined) {
180
+ if (auth.validate === undefined) return false;
181
+ if (await auth.validate(context)) {
182
+ args.logger.info('scrape.session.reused', { reused: true });
183
+ return false;
184
+ }
185
+ args.logger.info('scrape.session.expired', { reused: false });
186
+ await burnSession(args);
187
+ }
188
+ if (auth.login === undefined) throw sessionExpired(args.scrape, args.key);
189
+ await auth.login(context);
190
+ return true;
191
+ }
192
+
193
+ /** After a login: keep what it produced. A 2FA code somebody typed is worth exactly this. */
194
+ export async function persistSession<I>(plan: AuthPlanInput<I>, page: ScrapePage): Promise<void> {
195
+ const store = plan.auth?.store;
196
+ if (store === undefined) return;
197
+ const snapshot = await page.session();
198
+ await store.save({ ...snapshot, key: plan.key, savedAt: plan.clock.now().toISOString() });
199
+ plan.logger.info('scrape.session.saved', sessionDigest(snapshot));
200
+ }