@ultimat3/scraping 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +194 -0
- package/package.json +38 -0
- package/src/actionability.ts +106 -0
- package/src/artifacts.ts +69 -0
- package/src/auth.ts +200 -0
- package/src/cdp-fake.ts +150 -0
- package/src/cdp-port.ts +75 -0
- package/src/cdp-snapshot.ts +63 -0
- package/src/cdp-target.ts +320 -0
- package/src/clock.ts +97 -0
- package/src/cookie-scope.ts +97 -0
- package/src/driver-cdp.ts +184 -0
- package/src/driver-fake.ts +119 -0
- package/src/driver-fixture.ts +65 -0
- package/src/driver.ts +88 -0
- package/src/error-throws.ts +258 -0
- package/src/errors.ts +180 -0
- package/src/events.ts +74 -0
- package/src/expect.ts +133 -0
- package/src/failures.ts +47 -0
- package/src/hosts.ts +56 -0
- package/src/html-query.ts +119 -0
- package/src/html-requests.ts +45 -0
- package/src/html-target.ts +229 -0
- package/src/http-recorded.ts +85 -0
- package/src/http.ts +158 -0
- package/src/index.ts +178 -0
- package/src/intercept.ts +41 -0
- package/src/offline-session.ts +67 -0
- package/src/page-over-target.ts +215 -0
- package/src/page.ts +103 -0
- package/src/rate.ts +23 -0
- package/src/recording.ts +68 -0
- package/src/recover.ts +51 -0
- package/src/rings.ts +73 -0
- package/src/robots.ts +144 -0
- package/src/scrape-run.ts +226 -0
- package/src/scrape.ts +151 -0
- package/src/secrets.ts +91 -0
- package/src/session-state.ts +181 -0
- package/src/target.ts +118 -0
- package/src/watchdog.ts +100 -0
package/src/http.ts
ADDED
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
// The second transport, and the one that makes a scraper fast: drive the BROWSER through login,
|
|
2
|
+
// 2FA and navigation, then reverse-engineer the site's own JSON endpoints and pull the bulk over
|
|
3
|
+
// plain HTTP. Two hundred paginated pages clicked through is minutes and a hundred chances to
|
|
4
|
+
// break; the same data off the endpoint behind them is seconds, and a JSON endpoint changes far
|
|
5
|
+
// less often than a DOM.
|
|
6
|
+
//
|
|
7
|
+
// It is SESSION-BOUND, never a bare `fetch`. The browser's cookies, the browser's headers, the
|
|
8
|
+
// browser's proxy, the same `allowHosts`, the same robots gate, the same rate limit, the same
|
|
9
|
+
// cancellation. A second transport that quietly had none of those would be a hole in every
|
|
10
|
+
// guarantee the page vocabulary makes — and a different exit IP mid-session is exactly what
|
|
11
|
+
// anti-bot systems look for.
|
|
12
|
+
|
|
13
|
+
import type { StandardSchemaV1 } from '@ultimat3/schema';
|
|
14
|
+
import { parse } from '@ultimat3/schema';
|
|
15
|
+
import type { ScrapeClock } from './clock';
|
|
16
|
+
import { cookieHeaderFor } from './cookie-scope';
|
|
17
|
+
import { hostBlocked, httpFailed, scrapeTimeout } from './error-throws';
|
|
18
|
+
import type { InterceptRules } from './intercept';
|
|
19
|
+
import { interceptVerdict } from './intercept';
|
|
20
|
+
import type { NetworkRing } from './rings';
|
|
21
|
+
import type { RobotsGate } from './robots';
|
|
22
|
+
import type { SessionSnapshot } from './session-state';
|
|
23
|
+
|
|
24
|
+
export interface HttpRequestInit {
|
|
25
|
+
readonly method?: string | undefined;
|
|
26
|
+
readonly headers?: Readonly<Record<string, string>> | undefined;
|
|
27
|
+
readonly body?: string | undefined;
|
|
28
|
+
/** Milliseconds. Falls back to the session's own default. */
|
|
29
|
+
readonly timeout?: number | undefined;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
export interface ScrapeResponse {
|
|
33
|
+
readonly url: string;
|
|
34
|
+
readonly status: number;
|
|
35
|
+
readonly ok: boolean;
|
|
36
|
+
readonly headers: Readonly<Record<string, string>>;
|
|
37
|
+
text(): Promise<string>;
|
|
38
|
+
/** `unknown`, always. A response body is somebody else's JSON until a schema says otherwise. */
|
|
39
|
+
json(): Promise<unknown>;
|
|
40
|
+
/**
|
|
41
|
+
* Parse-or-throw, and the blessed path: a non-2xx answer is `X_SCRAPE_HTTP_FAILED` before the
|
|
42
|
+
* schema ever runs, so "the endpoint moved" never arrives as "the schema is wrong".
|
|
43
|
+
*/
|
|
44
|
+
parse<T>(schema: StandardSchemaV1<unknown, T>): Promise<T>;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
export interface ScrapeHttp {
|
|
48
|
+
/** One method. `get`/`post` sugar would be a second way to do the same thing (axiom 1). */
|
|
49
|
+
request(url: string, init?: HttpRequestInit): Promise<ScrapeResponse>;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
export interface HttpTransportInit {
|
|
53
|
+
readonly rules: InterceptRules;
|
|
54
|
+
readonly clock: ScrapeClock;
|
|
55
|
+
readonly timeoutMs: number;
|
|
56
|
+
readonly network: NetworkRing;
|
|
57
|
+
/**
|
|
58
|
+
* Read fresh on every request, and asynchronously because reading a real browser's jar is a
|
|
59
|
+
* round trip. A snapshot captured when the session opened would be the LOGGED-OUT one forever,
|
|
60
|
+
* which is precisely the handoff this transport exists to make.
|
|
61
|
+
*/
|
|
62
|
+
session(): Promise<SessionSnapshot>;
|
|
63
|
+
readonly robots?: RobotsGate | undefined;
|
|
64
|
+
readonly pace?: ((signal?: AbortSignal) => Promise<void>) | undefined;
|
|
65
|
+
readonly signal?: AbortSignal | undefined;
|
|
66
|
+
readonly onActivity?: (() => void) | undefined;
|
|
67
|
+
/** The SAME proxy the browser dialled through. A different exit IP is a different client. */
|
|
68
|
+
readonly proxy?: string | undefined;
|
|
69
|
+
readonly fetch?: typeof fetch | undefined;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
const headerRecord = (headers: Headers): Record<string, string> => {
|
|
73
|
+
const out: Record<string, string> = {};
|
|
74
|
+
headers.forEach((value, key) => {
|
|
75
|
+
out[key] = value;
|
|
76
|
+
});
|
|
77
|
+
return out;
|
|
78
|
+
};
|
|
79
|
+
|
|
80
|
+
export function responseOver(
|
|
81
|
+
url: string,
|
|
82
|
+
status: number,
|
|
83
|
+
headers: Readonly<Record<string, string>>,
|
|
84
|
+
body: () => Promise<string>,
|
|
85
|
+
): ScrapeResponse {
|
|
86
|
+
const ok = status >= 200 && status < 300;
|
|
87
|
+
const text = body;
|
|
88
|
+
return {
|
|
89
|
+
url,
|
|
90
|
+
status,
|
|
91
|
+
ok,
|
|
92
|
+
headers,
|
|
93
|
+
text,
|
|
94
|
+
json: async (): Promise<unknown> => JSON.parse(await text()) as unknown,
|
|
95
|
+
async parse<T>(schema: StandardSchemaV1<unknown, T>): Promise<T> {
|
|
96
|
+
if (!ok) throw httpFailed(url, status, (await text()).slice(0, 200));
|
|
97
|
+
return parse(schema, JSON.parse(await text()) as unknown);
|
|
98
|
+
},
|
|
99
|
+
};
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* The real transport. Every guarantee the page makes is re-applied here, in the same order and
|
|
104
|
+
* through the same functions — `interceptVerdict` is the one host rule, `RobotsGate` is the one
|
|
105
|
+
* robots rule, and neither is re-implemented for the second leg.
|
|
106
|
+
*/
|
|
107
|
+
export function httpOverFetch(init: HttpTransportInit): ScrapeHttp {
|
|
108
|
+
const call = init.fetch ?? fetch;
|
|
109
|
+
return {
|
|
110
|
+
async request(url: string, request: HttpRequestInit = {}): Promise<ScrapeResponse> {
|
|
111
|
+
init.onActivity?.();
|
|
112
|
+
if (interceptVerdict(url, 'fetch', init.rules) !== 'allow') {
|
|
113
|
+
throw hostBlocked(url, init.rules.allowHosts);
|
|
114
|
+
}
|
|
115
|
+
await init.robots?.assertAllowed(url);
|
|
116
|
+
await init.pace?.(init.signal);
|
|
117
|
+
const session = await init.session();
|
|
118
|
+
const cookies = cookieHeaderFor(session.cookies, url);
|
|
119
|
+
const timeoutMs = request.timeout ?? init.timeoutMs;
|
|
120
|
+
// `AbortSignal.timeout` and NOT `clock.sleep`: this is a deadline handed to the platform's
|
|
121
|
+
// own fetch, not a wait this package performs — and under a test clock a slept deadline
|
|
122
|
+
// would fire on the microtask after it was armed, cancelling every request instantly.
|
|
123
|
+
// The offline transport (`http-recorded.ts`) is what a test runs, and it has no deadline.
|
|
124
|
+
const deadlineSignal = AbortSignal.timeout(timeoutMs);
|
|
125
|
+
const signals = init.signal === undefined ? [deadlineSignal] : [deadlineSignal, init.signal];
|
|
126
|
+
try {
|
|
127
|
+
const response = await call(url, {
|
|
128
|
+
method: request.method ?? 'GET',
|
|
129
|
+
headers: {
|
|
130
|
+
...session.headers,
|
|
131
|
+
...(session.userAgent === '' ? {} : { 'user-agent': session.userAgent }),
|
|
132
|
+
...(cookies === undefined ? {} : { cookie: cookies }),
|
|
133
|
+
...request.headers,
|
|
134
|
+
},
|
|
135
|
+
...(request.body === undefined ? {} : { body: request.body }),
|
|
136
|
+
signal: AbortSignal.any(signals),
|
|
137
|
+
...(init.proxy === undefined ? {} : { proxy: init.proxy }),
|
|
138
|
+
} as RequestInit);
|
|
139
|
+
init.network.push({
|
|
140
|
+
method: request.method ?? 'GET',
|
|
141
|
+
url,
|
|
142
|
+
status: response.status,
|
|
143
|
+
resourceType: 'fetch',
|
|
144
|
+
at: init.clock.now().getTime(),
|
|
145
|
+
});
|
|
146
|
+
const body = await response.text();
|
|
147
|
+
return responseOver(url, response.status, headerRecord(response.headers), () =>
|
|
148
|
+
Promise.resolve(body),
|
|
149
|
+
);
|
|
150
|
+
} catch (thrown) {
|
|
151
|
+
// A deadline that fired is this package's own timeout, with its own code and fix — never
|
|
152
|
+
// the platform's bare `TimeoutError` reaching a job's retry classifier unclassified.
|
|
153
|
+
if (deadlineSignal.aborted) throw scrapeTimeout(`http ${url}`, timeoutMs);
|
|
154
|
+
throw thrown;
|
|
155
|
+
}
|
|
156
|
+
},
|
|
157
|
+
};
|
|
158
|
+
}
|
package/src/index.ts
ADDED
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
// Public API of @ultimat3/scraping. Explicit re-exports only — and complete enough that a third
|
|
2
|
+
// party can implement `ScrapeDriver` from this list alone. If a driver author needs a deep import,
|
|
3
|
+
// the seam is not a seam.
|
|
4
|
+
|
|
5
|
+
export type { ActionabilityState, ActionabilityWait } from './actionability';
|
|
6
|
+
export { actionabilityProblem, awaitActionable, DEFAULT_POLL_MS, isStable } from './actionability';
|
|
7
|
+
export type { ArtifactRef, ArtifactWriter, ArtifactWriterInit } from './artifacts';
|
|
8
|
+
export { contentTypeFor, createArtifactWriter, DEFAULT_ARTIFACT_PREFIX } from './artifacts';
|
|
9
|
+
export type {
|
|
10
|
+
AuthContext,
|
|
11
|
+
PromptHandler,
|
|
12
|
+
PromptRequest,
|
|
13
|
+
ScrapeAuth,
|
|
14
|
+
} from './auth';
|
|
15
|
+
export { burnSession, createPrompt, ensureAuthenticated, restorableSession } from './auth';
|
|
16
|
+
export type {
|
|
17
|
+
CdpBrowserLike,
|
|
18
|
+
CdpFrameLike,
|
|
19
|
+
CdpLauncherLike,
|
|
20
|
+
CdpPageLike,
|
|
21
|
+
CdpRequestLike,
|
|
22
|
+
} from './cdp-port';
|
|
23
|
+
export { parseSnapshots, snapshotExpression } from './cdp-snapshot';
|
|
24
|
+
export type { CdpTargetInit } from './cdp-target';
|
|
25
|
+
export { CDP_DRIVER, cdpTarget } from './cdp-target';
|
|
26
|
+
export type { Deadline, ScrapeClock, TestScrapeClock } from './clock';
|
|
27
|
+
export { deadline, systemScrapeClock, testClock, throwIfAborted } from './clock';
|
|
28
|
+
export {
|
|
29
|
+
cookieDomainMatches,
|
|
30
|
+
cookieHeaderFor,
|
|
31
|
+
cookiePathMatches,
|
|
32
|
+
cookiesForUrl,
|
|
33
|
+
} from './cookie-scope';
|
|
34
|
+
export type { ScrapeDriver, ScrapeSession, SessionInit } from './driver';
|
|
35
|
+
export { resetScrapeDriver, scrapeDriver, setScrapeDriver } from './driver';
|
|
36
|
+
export type { BrowserOptions, LocalBrowserOptions, RemoteBrowserOptions } from './driver-cdp';
|
|
37
|
+
export { localBrowser, remoteBrowser } from './driver-cdp';
|
|
38
|
+
export type { FakeBrowserOptions, FakePageOptions, FakePages } from './driver-fake';
|
|
39
|
+
export { FAKE_DRIVER, FAKE_PAGE_URL, fakeBrowser, fakePage, recordingsOf } from './driver-fake';
|
|
40
|
+
export type { FixtureBrowserOptions } from './driver-fixture';
|
|
41
|
+
export { FIXTURE_DRIVER, fixtureBrowser, recordingFilename } from './driver-fixture';
|
|
42
|
+
export {
|
|
43
|
+
authFailed,
|
|
44
|
+
blocked,
|
|
45
|
+
browserUnreachable,
|
|
46
|
+
cdpAttachFailed,
|
|
47
|
+
downloadTimeout,
|
|
48
|
+
driverUnknown,
|
|
49
|
+
fixtureMissing,
|
|
50
|
+
fixtureStale,
|
|
51
|
+
hostBlocked,
|
|
52
|
+
httpFailed,
|
|
53
|
+
notActionable,
|
|
54
|
+
outputInvalid,
|
|
55
|
+
pageCrashed,
|
|
56
|
+
profileLocked,
|
|
57
|
+
promptUnanswered,
|
|
58
|
+
recoverRefused,
|
|
59
|
+
remoteRequired,
|
|
60
|
+
robotsDisallowed,
|
|
61
|
+
scrapeNotImplemented,
|
|
62
|
+
scrapeTimeout,
|
|
63
|
+
secretExposed,
|
|
64
|
+
selectorMissing,
|
|
65
|
+
sessionExpired,
|
|
66
|
+
wedged,
|
|
67
|
+
yieldCollapsed,
|
|
68
|
+
} from './error-throws';
|
|
69
|
+
export type { ScrapeErrorCode, ScrapeErrorInit, ScrapeOwnedErrorCode } from './errors';
|
|
70
|
+
export {
|
|
71
|
+
isRetryableScrapeError,
|
|
72
|
+
isScrapeError,
|
|
73
|
+
SCRAPE_BORROWED_ERROR_CODES,
|
|
74
|
+
SCRAPE_ERROR_CODES,
|
|
75
|
+
SCRAPE_ERROR_RETRY,
|
|
76
|
+
SCRAPE_ERROR_TITLES,
|
|
77
|
+
SCRAPE_OWNED_ERROR_CODES,
|
|
78
|
+
ScrapeError,
|
|
79
|
+
} from './errors';
|
|
80
|
+
export type { ScrapeEventFields, StepEvent } from './events';
|
|
81
|
+
export { scrapeLogger, withStepEvent } from './events';
|
|
82
|
+
export type { YieldCheck, YieldExpectation, YieldGuardInput, YieldHistory } from './expect';
|
|
83
|
+
export {
|
|
84
|
+
DEFAULT_YIELD_WINDOW,
|
|
85
|
+
guardYield,
|
|
86
|
+
MIN_BASELINE_RUNS,
|
|
87
|
+
median,
|
|
88
|
+
memoryYieldHistory,
|
|
89
|
+
yieldProblem,
|
|
90
|
+
} from './expect';
|
|
91
|
+
export { BURNS_SESSION, burnsSession, errorCode, NEVER_RETRIED, neverRetried } from './failures';
|
|
92
|
+
export type { HostDecision, HostRule } from './hosts';
|
|
93
|
+
export { ANY_HOST, hostDecision, hostMatches } from './hosts';
|
|
94
|
+
export { markupEnabled, markupVisible, queryHtml } from './html-query';
|
|
95
|
+
export type { MarkupRequest } from './html-requests';
|
|
96
|
+
export { markupRequests } from './html-requests';
|
|
97
|
+
export type { HtmlTargetInit, RecordingLookup } from './html-target';
|
|
98
|
+
export { htmlTarget } from './html-target';
|
|
99
|
+
export type { HttpRequestInit, HttpTransportInit, ScrapeHttp, ScrapeResponse } from './http';
|
|
100
|
+
export { httpOverFetch, responseOver } from './http';
|
|
101
|
+
export type { HttpRecordingLookup, RecordedHttpInit } from './http-recorded';
|
|
102
|
+
export { httpRecordingFilename, httpRecordingsOf, recordedHttp } from './http-recorded';
|
|
103
|
+
export type { InterceptRules, InterceptVerdict } from './intercept';
|
|
104
|
+
export { interceptVerdict, refusalEntry } from './intercept';
|
|
105
|
+
export type { OfflineSessionInit } from './offline-session';
|
|
106
|
+
export { openOfflineSession } from './offline-session';
|
|
107
|
+
export type {
|
|
108
|
+
CaptureRequest,
|
|
109
|
+
DownloadRequest,
|
|
110
|
+
ElementValue,
|
|
111
|
+
ScrapeFrame,
|
|
112
|
+
ScrapePage,
|
|
113
|
+
WaitOptions,
|
|
114
|
+
} from './page';
|
|
115
|
+
export type { PageContext } from './page-over-target';
|
|
116
|
+
export { pageOverTarget } from './page-over-target';
|
|
117
|
+
export type { Pacer } from './rate';
|
|
118
|
+
export { createPacer, DEFAULT_NAVIGATION_RATE } from './rate';
|
|
119
|
+
export type { HttpRecording, PageRecording } from './recording';
|
|
120
|
+
export {
|
|
121
|
+
httpRecordingSchema,
|
|
122
|
+
pageRecordingSchema,
|
|
123
|
+
parseHttpRecording,
|
|
124
|
+
parseRecording,
|
|
125
|
+
splitDownload,
|
|
126
|
+
} from './recording';
|
|
127
|
+
export type { AgentRecovery, Recovery, RecoveryAttempt, RecoveryHook } from './recover';
|
|
128
|
+
export { runRecovery } from './recover';
|
|
129
|
+
export type {
|
|
130
|
+
ConsoleLine,
|
|
131
|
+
ConsoleRing,
|
|
132
|
+
NetworkEntry,
|
|
133
|
+
NetworkRing,
|
|
134
|
+
ResourceType,
|
|
135
|
+
Ring,
|
|
136
|
+
} from './rings';
|
|
137
|
+
export { createRing, DEFAULT_RING_CAPACITY, RESOURCE_TYPES } from './rings';
|
|
138
|
+
export type { RobotsFetch, RobotsGate, RobotsGateInit, RobotsPolicy, RobotsRules } from './robots';
|
|
139
|
+
export { createRobotsGate, DEFAULT_ROBOTS_AGENT, parseRobots, robotsAllows } from './robots';
|
|
140
|
+
export type {
|
|
141
|
+
ScrapeArtifacts,
|
|
142
|
+
ScrapeDefinition,
|
|
143
|
+
ScrapeReport,
|
|
144
|
+
ScrapeRunArgs,
|
|
145
|
+
} from './scrape';
|
|
146
|
+
export { scrape } from './scrape';
|
|
147
|
+
export { DEFAULT_PAGE_TIMEOUT_MS, runScrape } from './scrape-run';
|
|
148
|
+
export type { ScrapeSecrets, SecretResolver } from './secrets';
|
|
149
|
+
export {
|
|
150
|
+
blankPasswordFields,
|
|
151
|
+
createSecretBag,
|
|
152
|
+
redactSecrets,
|
|
153
|
+
SECRET_PLACEHOLDER,
|
|
154
|
+
safeHtml,
|
|
155
|
+
} from './secrets';
|
|
156
|
+
export type { ScrapeSessionStore, SessionSnapshot, SessionState } from './session-state';
|
|
157
|
+
export {
|
|
158
|
+
DEFAULT_SESSION_PREFIX,
|
|
159
|
+
EMPTY_SESSION,
|
|
160
|
+
memorySessionStore,
|
|
161
|
+
parseSessionState,
|
|
162
|
+
sessionDigest,
|
|
163
|
+
sessionKeyFor,
|
|
164
|
+
storageSessionStore,
|
|
165
|
+
} from './session-state';
|
|
166
|
+
export type {
|
|
167
|
+
CaptureOptions,
|
|
168
|
+
ElementBox,
|
|
169
|
+
ElementSnapshot,
|
|
170
|
+
FrameRef,
|
|
171
|
+
GotoOptions,
|
|
172
|
+
ScrapeCookie,
|
|
173
|
+
ScrapeDownloadFile,
|
|
174
|
+
ScrapeTarget,
|
|
175
|
+
} from './target';
|
|
176
|
+
export { ROOT_SELECTOR } from './target';
|
|
177
|
+
export type { WedgeGuard, WedgeGuardInit } from './watchdog';
|
|
178
|
+
export { createWedgeGuard, DEFAULT_GRACE_MS, DEFAULT_IDLE_MS } from './watchdog';
|
package/src/intercept.ts
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
// One decision, asked by every driver before a request leaves: allow it, refuse it because
|
|
2
|
+
// `allowHosts` does not list the host, or refuse it because `block` names the resource type.
|
|
3
|
+
//
|
|
4
|
+
// It lives in its own file so the fake, the fixture and the real browser cannot answer it
|
|
5
|
+
// differently — `driver-parity.test.ts` asserts the same refusal on all three, and a rule
|
|
6
|
+
// re-implemented per driver is a rule that holds only on the driver nobody ships.
|
|
7
|
+
|
|
8
|
+
import type { HostRule } from './hosts';
|
|
9
|
+
import { hostDecision } from './hosts';
|
|
10
|
+
import type { NetworkEntry, ResourceType } from './rings';
|
|
11
|
+
|
|
12
|
+
export interface InterceptRules {
|
|
13
|
+
readonly allowHosts: readonly HostRule[];
|
|
14
|
+
/** Resource types never fetched. `['image', 'media', 'font']` is most scrapes' whole config. */
|
|
15
|
+
readonly block?: readonly ResourceType[] | undefined;
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
export type InterceptVerdict = 'allow' | 'host' | 'blocked';
|
|
19
|
+
|
|
20
|
+
export function interceptVerdict(
|
|
21
|
+
url: string,
|
|
22
|
+
resourceType: ResourceType,
|
|
23
|
+
rules: InterceptRules,
|
|
24
|
+
): InterceptVerdict {
|
|
25
|
+
// Type first: a blocked image on an allowed host and a blocked image on a foreign host are both
|
|
26
|
+
// "never fetched", and reporting the cheaper reason keeps `allowHosts` findings meaningful.
|
|
27
|
+
if (rules.block?.includes(resourceType) === true) return 'blocked';
|
|
28
|
+
return hostDecision(url, rules.allowHosts).allowed ? 'allow' : 'host';
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/** The ring entry a refusal earns. Refusals are RECORDED, never silent — a scrape that came back
|
|
32
|
+
* empty is diagnosed from this list, and a blocked POST reported as a GET sends its reader hunting
|
|
33
|
+
* for a request the page never made. `method` is last and optional because a driver that cannot
|
|
34
|
+
* read one (a parsed document's `<img src>` is a GET by construction) says so by omitting it. */
|
|
35
|
+
export const refusalEntry = (
|
|
36
|
+
url: string,
|
|
37
|
+
resourceType: ResourceType,
|
|
38
|
+
verdict: Exclude<InterceptVerdict, 'allow'>,
|
|
39
|
+
at: number,
|
|
40
|
+
method = 'GET',
|
|
41
|
+
): NetworkEntry => ({ method, url, resourceType, at, refused: verdict });
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
// One assembly for both offline drivers: the recorded page target, the recorded HTTP transport
|
|
2
|
+
// and the page vocabulary over them. `fakeBrowser()` and `fixtureBrowser()` differ only in where
|
|
3
|
+
// the recordings come from — and a second copy of this wiring is how the two would drift apart on
|
|
4
|
+
// the day one of them learns about sessions and the other does not.
|
|
5
|
+
|
|
6
|
+
import type { ScrapeSession, SessionInit } from './driver';
|
|
7
|
+
import type { RecordingLookup } from './html-target';
|
|
8
|
+
import { htmlTarget } from './html-target';
|
|
9
|
+
import type { HttpRecordingLookup } from './http-recorded';
|
|
10
|
+
import { recordedHttp } from './http-recorded';
|
|
11
|
+
import { pageOverTarget } from './page-over-target';
|
|
12
|
+
import type { PageRecording } from './recording';
|
|
13
|
+
import type { SessionSnapshot } from './session-state';
|
|
14
|
+
import type { ScrapeCookie } from './target';
|
|
15
|
+
|
|
16
|
+
export interface OfflineSessionInit {
|
|
17
|
+
readonly driver: string;
|
|
18
|
+
readonly source: string;
|
|
19
|
+
readonly lookup: RecordingLookup;
|
|
20
|
+
readonly http: HttpRecordingLookup;
|
|
21
|
+
readonly session: SessionInit;
|
|
22
|
+
readonly start?: PageRecording | undefined;
|
|
23
|
+
readonly maxAgeMs?: number | undefined;
|
|
24
|
+
readonly cookies?: readonly ScrapeCookie[] | undefined;
|
|
25
|
+
readonly snapshot?: SessionSnapshot | undefined;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
export async function openOfflineSession(init: OfflineSessionInit): Promise<ScrapeSession> {
|
|
29
|
+
const target = htmlTarget({
|
|
30
|
+
driver: init.driver,
|
|
31
|
+
lookup: init.lookup,
|
|
32
|
+
rules: init.session.rules,
|
|
33
|
+
clock: init.session.clock,
|
|
34
|
+
source: init.source,
|
|
35
|
+
start: init.start,
|
|
36
|
+
maxAgeMs: init.maxAgeMs,
|
|
37
|
+
cookies: init.cookies,
|
|
38
|
+
session: init.snapshot,
|
|
39
|
+
});
|
|
40
|
+
// Restored BEFORE the first navigation, exactly as a real driver must: a session put back after
|
|
41
|
+
// the first request is a first request made logged out.
|
|
42
|
+
if (init.session.restore !== undefined) await target.restore(init.session.restore);
|
|
43
|
+
return {
|
|
44
|
+
driver: init.driver,
|
|
45
|
+
page: pageOverTarget(target, {
|
|
46
|
+
clock: init.session.clock,
|
|
47
|
+
allowHosts: init.session.rules.allowHosts,
|
|
48
|
+
defaultTimeoutMs: init.session.timeoutMs,
|
|
49
|
+
secrets: init.session.secrets,
|
|
50
|
+
robots: init.session.robots,
|
|
51
|
+
signal: init.session.signal,
|
|
52
|
+
onActivity: init.session.onActivity,
|
|
53
|
+
pace: init.session.pace,
|
|
54
|
+
}),
|
|
55
|
+
http: recordedHttp({
|
|
56
|
+
lookup: init.http,
|
|
57
|
+
rules: init.session.rules,
|
|
58
|
+
network: target.network,
|
|
59
|
+
clock: init.session.clock,
|
|
60
|
+
source: init.source,
|
|
61
|
+
// The same gate the page above holds, from the same field: two legs, one robots decision.
|
|
62
|
+
robots: init.session.robots,
|
|
63
|
+
maxAgeMs: init.maxAgeMs,
|
|
64
|
+
}),
|
|
65
|
+
close: () => target.close(),
|
|
66
|
+
};
|
|
67
|
+
}
|
|
@@ -0,0 +1,215 @@
|
|
|
1
|
+
// The ONE implementation of the page vocabulary, over the `ScrapeTarget` port. Real browser,
|
|
2
|
+
// recorded fixture and parsed HTML all reach `run()` through this file — so actionability, frame
|
|
3
|
+
// re-resolution, host enforcement and secret taint are written once and cannot drift between the
|
|
4
|
+
// driver a test uses and the driver production uses.
|
|
5
|
+
|
|
6
|
+
import type { Secret } from '@ultimat3/core';
|
|
7
|
+
import { isSecret, revealSecret } from '@ultimat3/core';
|
|
8
|
+
import type { ActionabilityState } from './actionability';
|
|
9
|
+
import { awaitActionable } from './actionability';
|
|
10
|
+
import type { ScrapeClock } from './clock';
|
|
11
|
+
import { deadline } from './clock';
|
|
12
|
+
import { hostBlocked, secretExposed, selectorMissing } from './error-throws';
|
|
13
|
+
import { hostDecision } from './hosts';
|
|
14
|
+
import type {
|
|
15
|
+
CaptureRequest,
|
|
16
|
+
DownloadRequest,
|
|
17
|
+
ElementValue,
|
|
18
|
+
ScrapeFrame,
|
|
19
|
+
ScrapePage,
|
|
20
|
+
WaitOptions,
|
|
21
|
+
} from './page';
|
|
22
|
+
import type { RobotsGate } from './robots';
|
|
23
|
+
import type { ScrapeSecrets } from './secrets';
|
|
24
|
+
import { safeHtml } from './secrets';
|
|
25
|
+
import type { ElementSnapshot, ScrapeCookie, ScrapeDownloadFile, ScrapeTarget } from './target';
|
|
26
|
+
import { ROOT_SELECTOR } from './target';
|
|
27
|
+
|
|
28
|
+
export interface PageContext {
|
|
29
|
+
readonly clock: ScrapeClock;
|
|
30
|
+
/**
|
|
31
|
+
* Called before EVERY operation, poll included. This is what the wedge watchdog measures: the
|
|
32
|
+
* gap between two of these is the definition of "the browser stopped answering", and putting
|
|
33
|
+
* the hook here rather than in a wrapper means no page method can forget to report.
|
|
34
|
+
*/
|
|
35
|
+
readonly onActivity?: (() => void) | undefined;
|
|
36
|
+
/** Awaited before each navigation. `rate.ts` builds it; omitted means unpaced. */
|
|
37
|
+
readonly pace?: ((signal?: AbortSignal) => Promise<void>) | undefined;
|
|
38
|
+
readonly allowHosts: readonly string[];
|
|
39
|
+
readonly defaultTimeoutMs: number;
|
|
40
|
+
readonly secrets?: ScrapeSecrets | undefined;
|
|
41
|
+
readonly robots?: RobotsGate | undefined;
|
|
42
|
+
readonly signal?: AbortSignal | undefined;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/** Mutable, shared by the page and every frame under it: a taint is a property of the SESSION. */
|
|
46
|
+
interface PageState {
|
|
47
|
+
tainted: boolean;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
type Resolve = () => Promise<ScrapeTarget>;
|
|
51
|
+
|
|
52
|
+
/** Every operation resolves its target through this, so every operation reports activity. */
|
|
53
|
+
const watched = (resolve: Resolve, ctx: PageContext): Resolve => {
|
|
54
|
+
return () => {
|
|
55
|
+
ctx.onActivity?.();
|
|
56
|
+
return resolve();
|
|
57
|
+
};
|
|
58
|
+
};
|
|
59
|
+
|
|
60
|
+
const plainText = (text: string | Secret): string => (isSecret(text) ? revealSecret(text) : text);
|
|
61
|
+
|
|
62
|
+
const toValue = (snapshot: ElementSnapshot): ElementValue => ({
|
|
63
|
+
tag: snapshot.tag,
|
|
64
|
+
text: snapshot.text,
|
|
65
|
+
value: snapshot.value,
|
|
66
|
+
attrs: snapshot.attrs,
|
|
67
|
+
});
|
|
68
|
+
|
|
69
|
+
async function first(resolve: Resolve, selector: string): Promise<ElementSnapshot | undefined> {
|
|
70
|
+
const target = await resolve();
|
|
71
|
+
return (await target.query(selector))[0];
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
function frameOver(
|
|
75
|
+
rawResolve: Resolve,
|
|
76
|
+
ctx: PageContext,
|
|
77
|
+
state: PageState,
|
|
78
|
+
seed: string,
|
|
79
|
+
): ScrapeFrame {
|
|
80
|
+
const resolve = watched(rawResolve, ctx);
|
|
81
|
+
let lastUrl = seed;
|
|
82
|
+
const timeoutFor = (options: WaitOptions | undefined): number =>
|
|
83
|
+
options?.timeout ?? ctx.defaultTimeoutMs;
|
|
84
|
+
|
|
85
|
+
const wait = async (
|
|
86
|
+
selector: string,
|
|
87
|
+
options: WaitOptions | undefined,
|
|
88
|
+
fallback: ActionabilityState,
|
|
89
|
+
): Promise<ElementSnapshot> => {
|
|
90
|
+
const target = await resolve();
|
|
91
|
+
lastUrl = target.url();
|
|
92
|
+
return awaitActionable({
|
|
93
|
+
selector,
|
|
94
|
+
url: lastUrl,
|
|
95
|
+
state: options?.state ?? fallback,
|
|
96
|
+
timeoutMs: timeoutFor(options),
|
|
97
|
+
clock: ctx.clock,
|
|
98
|
+
signal: ctx.signal,
|
|
99
|
+
snapshot: () => first(resolve, selector),
|
|
100
|
+
});
|
|
101
|
+
};
|
|
102
|
+
|
|
103
|
+
return {
|
|
104
|
+
url: () => lastUrl,
|
|
105
|
+
waitFor: (selector, options) => wait(selector, options, 'actionable'),
|
|
106
|
+
async click(selector, options): Promise<void> {
|
|
107
|
+
await wait(selector, options, 'actionable');
|
|
108
|
+
await (await resolve()).click(selector, 0);
|
|
109
|
+
},
|
|
110
|
+
async type(selector, text, options): Promise<void> {
|
|
111
|
+
await wait(selector, options, 'actionable');
|
|
112
|
+
if (isSecret(text)) state.tainted = true;
|
|
113
|
+
await (await resolve()).type(selector, plainText(text));
|
|
114
|
+
},
|
|
115
|
+
async fill(selector, text, options): Promise<void> {
|
|
116
|
+
await wait(selector, options, 'actionable');
|
|
117
|
+
const target = await resolve();
|
|
118
|
+
await target.clear(selector);
|
|
119
|
+
if (isSecret(text)) state.tainted = true;
|
|
120
|
+
await target.type(selector, plainText(text));
|
|
121
|
+
},
|
|
122
|
+
async select(selector, values, options): Promise<void> {
|
|
123
|
+
await wait(selector, options, 'actionable');
|
|
124
|
+
await (await resolve()).select(selector, values);
|
|
125
|
+
},
|
|
126
|
+
async values(selector): Promise<readonly ElementValue[]> {
|
|
127
|
+
return (await (await resolve()).query(selector)).map(toValue);
|
|
128
|
+
},
|
|
129
|
+
async text(selector): Promise<string> {
|
|
130
|
+
const target = await resolve();
|
|
131
|
+
if (selector === undefined) return (await target.query(ROOT_SELECTOR))[0]?.text ?? '';
|
|
132
|
+
return (await target.query(selector))[0]?.text ?? '';
|
|
133
|
+
},
|
|
134
|
+
async html(): Promise<string> {
|
|
135
|
+
return safeHtml(await (await resolve()).content(), ctx.secrets);
|
|
136
|
+
},
|
|
137
|
+
async count(selector): Promise<number> {
|
|
138
|
+
return (await (await resolve()).query(selector)).length;
|
|
139
|
+
},
|
|
140
|
+
async evaluate(expression): Promise<unknown> {
|
|
141
|
+
return (await resolve()).evaluate(expression);
|
|
142
|
+
},
|
|
143
|
+
frame(nameOrSelector): ScrapeFrame {
|
|
144
|
+
// The resolver, not the frame. Every call through the returned handle runs this again, so
|
|
145
|
+
// a handle taken before a re-navigation addresses the CURRENT frame with that name and
|
|
146
|
+
// never a detached one — the trap `frameLocator`-style handles set for every caller.
|
|
147
|
+
const resolveChild = async (): Promise<ScrapeTarget> => {
|
|
148
|
+
const budget = deadline(ctx.clock, ctx.defaultTimeoutMs);
|
|
149
|
+
for (;;) {
|
|
150
|
+
const parent = await resolve();
|
|
151
|
+
const found = (await parent.frames()).find(
|
|
152
|
+
(ref) =>
|
|
153
|
+
ref.name === nameOrSelector ||
|
|
154
|
+
ref.selector === nameOrSelector ||
|
|
155
|
+
ref.url === nameOrSelector,
|
|
156
|
+
);
|
|
157
|
+
if (found !== undefined) return found.target;
|
|
158
|
+
if (budget.expired()) {
|
|
159
|
+
throw selectorMissing(nameOrSelector, parent.url(), ctx.defaultTimeoutMs);
|
|
160
|
+
}
|
|
161
|
+
await ctx.clock.sleep(Math.min(50, budget.remainingMs()), ctx.signal);
|
|
162
|
+
}
|
|
163
|
+
};
|
|
164
|
+
return frameOver(resolveChild, ctx, state, lastUrl);
|
|
165
|
+
},
|
|
166
|
+
};
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
/**
|
|
170
|
+
* Refused BEFORE the navigation, not reported after it. An `allowHosts` consulted afterwards is a
|
|
171
|
+
* log line about a request that already left the container.
|
|
172
|
+
*/
|
|
173
|
+
async function guardNavigation(url: string, ctx: PageContext): Promise<void> {
|
|
174
|
+
const decision = hostDecision(url, ctx.allowHosts);
|
|
175
|
+
if (!decision.allowed) throw hostBlocked(url, ctx.allowHosts);
|
|
176
|
+
await ctx.robots?.assertAllowed(url);
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
export function pageOverTarget(target: ScrapeTarget, ctx: PageContext): ScrapePage {
|
|
180
|
+
const state: PageState = { tainted: false };
|
|
181
|
+
const frame = frameOver(() => Promise.resolve(target), ctx, state, target.url());
|
|
182
|
+
const capture = async (
|
|
183
|
+
kind: 'screenshot' | 'pdf',
|
|
184
|
+
options: CaptureRequest | undefined,
|
|
185
|
+
): Promise<Uint8Array> => {
|
|
186
|
+
// The leak nobody remembers: a screenshot of a filled login form IS the password, in pixels,
|
|
187
|
+
// in object storage, forever. Refused rather than masked — a mask over pixels is a guess
|
|
188
|
+
// about layout, and `page.html()` already gives a redacted artifact that is exact.
|
|
189
|
+
if (state.tainted) throw secretExposed(kind, target.url());
|
|
190
|
+
const timeoutMs = options?.timeout ?? ctx.defaultTimeoutMs;
|
|
191
|
+
const request = { fullPage: options?.fullPage, timeoutMs };
|
|
192
|
+
return kind === 'screenshot' ? target.screenshot(request) : target.pdf(request);
|
|
193
|
+
};
|
|
194
|
+
return {
|
|
195
|
+
...frame,
|
|
196
|
+
async goto(url, options): Promise<void> {
|
|
197
|
+
await guardNavigation(url, ctx);
|
|
198
|
+
await ctx.pace?.(ctx.signal);
|
|
199
|
+
ctx.onActivity?.();
|
|
200
|
+
await target.goto(url, {
|
|
201
|
+
timeoutMs: options?.timeout ?? ctx.defaultTimeoutMs,
|
|
202
|
+
signal: ctx.signal,
|
|
203
|
+
});
|
|
204
|
+
},
|
|
205
|
+
screenshot: (options) => capture('screenshot', options),
|
|
206
|
+
pdf: (options) => capture('pdf', options),
|
|
207
|
+
download: (options?: DownloadRequest): Promise<ScrapeDownloadFile> =>
|
|
208
|
+
target.download({ timeoutMs: options?.timeout ?? ctx.defaultTimeoutMs }),
|
|
209
|
+
cookies: (): Promise<readonly ScrapeCookie[]> => target.cookies(),
|
|
210
|
+
session: () => target.session(),
|
|
211
|
+
console: () => target.console.entries(),
|
|
212
|
+
network: () => target.network.entries(),
|
|
213
|
+
networkDropped: () => target.network.dropped,
|
|
214
|
+
};
|
|
215
|
+
}
|