@crawlee/playwright 4.0.0-beta.21 → 4.0.0-beta.210
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -14
- package/index.d.ts +4 -3
- package/index.js +2 -2
- package/internals/adaptive-playwright-crawler.d.ts +136 -65
- package/internals/adaptive-playwright-crawler.js +417 -274
- package/internals/enqueue-links/click-elements.d.ts +36 -64
- package/internals/enqueue-links/click-elements.js +65 -67
- package/internals/playwright-browser-pool.d.ts +71 -0
- package/internals/playwright-browser-pool.js +61 -0
- package/internals/playwright-crawler.d.ts +176 -148
- package/internals/playwright-crawler.js +77 -73
- package/internals/playwright-launcher.d.ts +30 -20
- package/internals/playwright-launcher.js +22 -17
- package/internals/utils/playwright-utils.d.ts +54 -56
- package/internals/utils/playwright-utils.js +117 -138
- package/internals/utils/rendering-type-prediction.d.ts +37 -11
- package/internals/utils/rendering-type-prediction.js +81 -27
- package/package.json +15 -19
- package/index.d.ts.map +0 -1
- package/index.js.map +0 -1
- package/internals/adaptive-playwright-crawler.d.ts.map +0 -1
- package/internals/adaptive-playwright-crawler.js.map +0 -1
- package/internals/enqueue-links/click-elements.d.ts.map +0 -1
- package/internals/enqueue-links/click-elements.js.map +0 -1
- package/internals/playwright-crawler.d.ts.map +0 -1
- package/internals/playwright-crawler.js.map +0 -1
- package/internals/playwright-launcher.d.ts.map +0 -1
- package/internals/playwright-launcher.js.map +0 -1
- package/internals/utils/playwright-utils.d.ts.map +0 -1
- package/internals/utils/playwright-utils.js.map +0 -1
- package/internals/utils/rendering-type-prediction.d.ts.map +0 -1
- package/internals/utils/rendering-type-prediction.js.map +0 -1
|
@@ -1,47 +1,30 @@
|
|
|
1
|
+
import { setTimeout as delay } from 'node:timers/promises';
|
|
1
2
|
import { isDeepStrictEqual } from 'node:util';
|
|
2
|
-
import { BasicCrawler } from '@crawlee/basic';
|
|
3
|
+
import { BasicCrawler, ContextPipeline, ContextPipelineInitializationError, RequestHandlerError, resolveBaseUrlForEnqueueLinksFiltering, Router, Statistics, } from '@crawlee/basic';
|
|
3
4
|
import { extractUrlsFromPage } from '@crawlee/browser';
|
|
4
5
|
import { CheerioCrawler } from '@crawlee/cheerio';
|
|
5
|
-
import {
|
|
6
|
-
import { extractUrlsFromCheerio } from '@crawlee/utils';
|
|
6
|
+
import { createStorageTransaction, EnqueueStrategy, OwnedOrInjected } from '@crawlee/core';
|
|
7
|
+
import { extractUrlsFromCheerio, parseArgument } from '@crawlee/utils/internal';
|
|
8
|
+
import { z } from 'zod';
|
|
7
9
|
import { addTimeoutToPromise } from '@apify/timeout';
|
|
8
10
|
import { PlaywrightCrawler } from './playwright-crawler.js';
|
|
9
|
-
import { RenderingTypePredictor } from './utils/rendering-type-prediction.js';
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
return;
|
|
27
|
-
}
|
|
28
|
-
this.state.httpOnlyRequestHandlerRuns = savedState.httpOnlyRequestHandlerRuns;
|
|
29
|
-
this.state.browserRequestHandlerRuns = savedState.browserRequestHandlerRuns;
|
|
30
|
-
this.state.renderingTypeMispredictions = savedState.renderingTypeMispredictions;
|
|
31
|
-
}
|
|
32
|
-
trackHttpOnlyRequestHandlerRun() {
|
|
33
|
-
this.state.httpOnlyRequestHandlerRuns ??= 0;
|
|
34
|
-
this.state.httpOnlyRequestHandlerRuns += 1;
|
|
35
|
-
}
|
|
36
|
-
trackBrowserRequestHandlerRun() {
|
|
37
|
-
this.state.browserRequestHandlerRuns ??= 0;
|
|
38
|
-
this.state.browserRequestHandlerRuns += 1;
|
|
39
|
-
}
|
|
40
|
-
trackRenderingTypeMisprediction() {
|
|
41
|
-
this.state.renderingTypeMispredictions ??= 0;
|
|
42
|
-
this.state.renderingTypeMispredictions += 1;
|
|
43
|
-
}
|
|
44
|
-
}
|
|
11
|
+
import { RenderingTypePredictor, } from './utils/rendering-type-prediction.js';
|
|
12
|
+
const adaptiveStatisticStateSchema = z.object({
|
|
13
|
+
/** How many requests were handled by the HTTP-only request handler. */
|
|
14
|
+
httpOnlyRequestHandlerRuns: z.number().default(0),
|
|
15
|
+
/** How many requests were handled in a browser. */
|
|
16
|
+
browserRequestHandlerRuns: z.number().default(0),
|
|
17
|
+
/** How many times the HTTP-only handler produced a result the `resultChecker` rejected. */
|
|
18
|
+
renderingTypeMispredictions: z.number().default(0),
|
|
19
|
+
});
|
|
20
|
+
/**
|
|
21
|
+
* The {@link AdaptivePlaywrightCrawlerStatisticState} fields as a {@link Statistics} state extension, defaults
|
|
22
|
+
* and all. A {@link Statistics} instance to be injected into an {@link AdaptivePlaywrightCrawler} has to carry
|
|
23
|
+
* them - `deserialize.extend()` your own fields onto this one and pass the result as `stateExtension`.
|
|
24
|
+
*/
|
|
25
|
+
export const adaptivePlaywrightCrawlerStatisticState = {
|
|
26
|
+
deserialize: adaptiveStatisticStateSchema,
|
|
27
|
+
};
|
|
45
28
|
const proxyLogMethods = [
|
|
46
29
|
'error',
|
|
47
30
|
'exception',
|
|
@@ -82,42 +65,91 @@ const proxyLogMethods = [
|
|
|
82
65
|
* @experimental
|
|
83
66
|
*/
|
|
84
67
|
export class AdaptivePlaywrightCrawler extends BasicCrawler {
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
resultComparator;
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
68
|
+
#renderingTypePredictor;
|
|
69
|
+
#resultChecker;
|
|
70
|
+
#shouldPropagateError;
|
|
71
|
+
#resultComparator;
|
|
72
|
+
#staticContextPipeline;
|
|
73
|
+
#browserContextPipeline;
|
|
74
|
+
#individualRequestHandlerTimeoutMillis;
|
|
75
|
+
/**
|
|
76
|
+
* The write policy of the per-attempt transactions. Defaults the request queue to `deferred`:
|
|
77
|
+
* a discarded attempt's enqueues must never reach the queue.
|
|
78
|
+
*/
|
|
79
|
+
#attemptWritePolicy;
|
|
80
|
+
/** Owns the browser pool this crawler's runs use, so its per-run resources are released with ours. */
|
|
81
|
+
#browserCrawler;
|
|
82
|
+
/** Nothing of its state is per-run, but it owns a session pool that outlives one. */
|
|
83
|
+
#staticCrawler;
|
|
84
|
+
/**
|
|
85
|
+
* In-flight rendering type detections, plus the pending results of an asynchronous `storeResult`.
|
|
86
|
+
*/
|
|
87
|
+
#activeDetections = new Set();
|
|
88
|
+
/**
|
|
89
|
+
* Set once `teardown()` starts, so that requests still in the pool stop opening new detections.
|
|
90
|
+
*/
|
|
91
|
+
#shutDown = false;
|
|
92
|
+
constructor(options = {}) {
|
|
93
|
+
const { requestHandler, renderingTypeDetectionRatio = 0.1, renderingTypePredictor, resultChecker, shouldPropagateError, resultComparator, statistics, requestHandlerTimeoutSecs = 60, errorHandler, failedRequestHandler, preNavigationHooks = [], postNavigationHooks = [], extendContext, transactionalStorage, launchContext, headless, browserPool, remoteBrowser, ...rest } = options;
|
|
94
|
+
// The user's value is replaced by `false` in the `super` call below — validate it separately,
|
|
95
|
+
// wrapped in an object so the error still names the field.
|
|
96
|
+
parseArgument({ transactionalStorage }, z.object({ transactionalStorage: BasicCrawler.optionsShape.transactionalStorage }), 'AdaptivePlaywrightCrawlerOptions');
|
|
97
|
+
// The extra fields are only tracked if the injected instance was built with them - the types enforce that,
|
|
98
|
+
// but plain JS callers would otherwise silently increment `undefined` into a sticky `NaN`. Extend
|
|
99
|
+
// `adaptivePlaywrightCrawlerStatisticState` to satisfy this.
|
|
100
|
+
if (statistics !== undefined) {
|
|
101
|
+
parseArgument(statistics.state, z.object({
|
|
102
|
+
httpOnlyRequestHandlerRuns: z.number(),
|
|
103
|
+
browserRequestHandlerRuns: z.number(),
|
|
104
|
+
renderingTypeMispredictions: z.number(),
|
|
105
|
+
}), 'statistics.state');
|
|
106
|
+
}
|
|
107
|
+
// Per-attempt buffering is load-bearing here: the handler runs up to twice per request and the
|
|
108
|
+
// losing attempt's writes must be discardable.
|
|
109
|
+
if (transactionalStorage === false) {
|
|
110
|
+
throw new Error('AdaptivePlaywrightCrawler requires transactional storage - it runs the request handler ' +
|
|
111
|
+
'multiple times per request and must be able to discard the storage writes of losing ' +
|
|
112
|
+
'attempts. `transactionalStorage: false` is therefore not supported; a write policy ' +
|
|
113
|
+
'object is accepted and forwarded to the per-attempt transactions.');
|
|
114
|
+
}
|
|
97
115
|
super({
|
|
98
116
|
...rest,
|
|
99
|
-
// Pass error handlers to the "main" crawler - we only pluck them from `rest` so that they don't go to the sub crawlers
|
|
100
117
|
errorHandler,
|
|
101
118
|
failedRequestHandler,
|
|
102
|
-
// Same for request handler
|
|
103
119
|
requestHandler,
|
|
104
|
-
|
|
105
|
-
//
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
120
|
+
requestHandlerTimeoutSecs,
|
|
121
|
+
// The base would build a `Statistics` without the adaptive fields, so provide a default that has them.
|
|
122
|
+
// The cast covers a `StatisticStateExtension` that adds further fields - those can only come from an
|
|
123
|
+
// injected instance, in which case this default is never built.
|
|
124
|
+
statistics: statistics ??
|
|
125
|
+
new Statistics({
|
|
126
|
+
logMessage: `${AdaptivePlaywrightCrawler.name} request statistics:`,
|
|
127
|
+
stateExtension: adaptivePlaywrightCrawlerStatisticState,
|
|
128
|
+
}),
|
|
129
|
+
contextPipelineBuilder: () => this.#buildContextPipeline(),
|
|
130
|
+
// The base crawler must not wrap requests in a transaction of its own - this crawler opens
|
|
131
|
+
// one per request handler attempt in `crawlOne` instead, forwarding the write policy of the
|
|
132
|
+
// user-facing option (validated above) to those.
|
|
133
|
+
transactionalStorage: false,
|
|
134
|
+
});
|
|
135
|
+
this.#individualRequestHandlerTimeoutMillis = requestHandlerTimeoutSecs * 1000;
|
|
136
|
+
// `renderingTypeDetectionRatio` only configures the default predictor - an injected one brings its own
|
|
137
|
+
// detection ratio (and its own state), so the option is ignored in that case.
|
|
138
|
+
this.#renderingTypePredictor = OwnedOrInjected.resolve(renderingTypePredictor, () => new RenderingTypePredictor({ detectionRatio: renderingTypeDetectionRatio }));
|
|
139
|
+
this.#attemptWritePolicy = {
|
|
140
|
+
requestQueue: 'deferred',
|
|
141
|
+
...(typeof transactionalStorage === 'object' ? transactionalStorage : {}),
|
|
142
|
+
};
|
|
143
|
+
this.#resultChecker = resultChecker ?? (() => true);
|
|
144
|
+
this.#shouldPropagateError = shouldPropagateError ?? (() => false);
|
|
113
145
|
if (resultComparator !== undefined) {
|
|
114
|
-
this
|
|
146
|
+
this.#resultComparator = resultComparator;
|
|
115
147
|
}
|
|
116
148
|
else if (resultChecker !== undefined) {
|
|
117
|
-
this
|
|
149
|
+
this.#resultComparator = (resultA, resultB) => this.#resultChecker(resultA) && this.#resultChecker(resultB);
|
|
118
150
|
}
|
|
119
151
|
else {
|
|
120
|
-
this
|
|
152
|
+
this.#resultComparator = (resultA, resultB) => {
|
|
121
153
|
return (resultA.datasetItems.length === resultB.datasetItems.length &&
|
|
122
154
|
resultA.datasetItems.every((itemA, i) => {
|
|
123
155
|
const itemB = resultB.datasetItems[i];
|
|
@@ -125,101 +157,95 @@ export class AdaptivePlaywrightCrawler extends BasicCrawler {
|
|
|
125
157
|
}));
|
|
126
158
|
};
|
|
127
159
|
}
|
|
160
|
+
// `extendContext` is forwarded to the inner crawlers, which run it *before* navigation (see
|
|
161
|
+
// `BasicCrawler`), keeping the behavior consistent with the non-adaptive crawlers: the
|
|
162
|
+
// extension is visible to the pre/post-navigation hooks and the request handler, but cannot
|
|
163
|
+
// access navigation-dependent members (`page`, `response`, `$`, ...).
|
|
164
|
+
//
|
|
165
|
+
// The adaptive hooks target a subset context (`AdaptiveHookContext`); the casts to the inner
|
|
166
|
+
// crawlers' `PlaywrightHook` type relax that nominal difference. The `ContextPipeline` merges
|
|
167
|
+
// each hook's overrides at runtime regardless of the static type.
|
|
128
168
|
const staticCrawler = new CheerioCrawler({
|
|
129
169
|
...rest,
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
async (context) => {
|
|
136
|
-
for (const hook of preNavigationHooks ?? []) {
|
|
137
|
-
await hook(context, undefined);
|
|
138
|
-
}
|
|
139
|
-
},
|
|
140
|
-
],
|
|
141
|
-
postNavigationHooks: [
|
|
142
|
-
async (context) => {
|
|
143
|
-
for (const hook of postNavigationHooks ?? []) {
|
|
144
|
-
await hook(context, undefined);
|
|
145
|
-
}
|
|
146
|
-
},
|
|
147
|
-
],
|
|
148
|
-
}, config);
|
|
170
|
+
statistics: new Statistics({ persistenceOptions: { enable: false } }),
|
|
171
|
+
preNavigationHooks,
|
|
172
|
+
postNavigationHooks,
|
|
173
|
+
extendContext,
|
|
174
|
+
});
|
|
149
175
|
const browserCrawler = new PlaywrightCrawler({
|
|
150
176
|
...rest,
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
}
|
|
160
|
-
},
|
|
161
|
-
],
|
|
162
|
-
postNavigationHooks: [
|
|
163
|
-
async (context, gotoOptions) => {
|
|
164
|
-
for (const hook of postNavigationHooks ?? []) {
|
|
165
|
-
await hook(context, gotoOptions);
|
|
166
|
-
}
|
|
167
|
-
},
|
|
168
|
-
],
|
|
169
|
-
}, config);
|
|
170
|
-
this.teardownHooks.push(browserCrawler.teardown.bind(browserCrawler));
|
|
171
|
-
this.staticContextPipeline = staticCrawler.contextPipeline
|
|
172
|
-
.compose({
|
|
173
|
-
action: this.adaptCheerioContext.bind(this),
|
|
174
|
-
})
|
|
175
|
-
.compose({
|
|
176
|
-
action: async (context) => extendContext ? await extendContext(context) : context,
|
|
177
|
-
});
|
|
178
|
-
this.browserContextPipeline = browserCrawler.contextPipeline
|
|
179
|
-
.compose({
|
|
180
|
-
action: this.adaptPlaywrightContext.bind(this),
|
|
181
|
-
})
|
|
182
|
-
.compose({
|
|
183
|
-
action: async (context) => extendContext ? await extendContext(context) : context,
|
|
177
|
+
statistics: new Statistics({ persistenceOptions: { enable: false } }),
|
|
178
|
+
preNavigationHooks: preNavigationHooks,
|
|
179
|
+
postNavigationHooks: postNavigationHooks,
|
|
180
|
+
extendContext,
|
|
181
|
+
launchContext,
|
|
182
|
+
headless,
|
|
183
|
+
browserPool,
|
|
184
|
+
remoteBrowser,
|
|
184
185
|
});
|
|
185
|
-
this
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
});
|
|
190
|
-
this.preventDirectStorageAccess = preventDirectStorageAccess;
|
|
186
|
+
this.#staticCrawler = staticCrawler;
|
|
187
|
+
this.#browserCrawler = browserCrawler;
|
|
188
|
+
this.#staticContextPipeline = staticCrawler.contextPipeline.compose(this.adaptCheerioContext.bind(this));
|
|
189
|
+
this.#browserContextPipeline = browserCrawler.contextPipeline.compose(this.adaptPlaywrightContext.bind(this));
|
|
191
190
|
}
|
|
192
|
-
async
|
|
193
|
-
|
|
194
|
-
|
|
191
|
+
async init() {
|
|
192
|
+
// A crawler can be run again after a teardown.
|
|
193
|
+
this.#shutDown = false;
|
|
194
|
+
// Only the predictor we built ourselves is ours to initialize - an injected one is borrowed, so its
|
|
195
|
+
// lifecycle (including restoring persisted state) stays with whoever created it.
|
|
196
|
+
await this.#renderingTypePredictor.ifOwned((predictor) => predictor.initialize());
|
|
197
|
+
return await super.init();
|
|
198
|
+
}
|
|
199
|
+
#buildContextPipeline() {
|
|
200
|
+
const errorMessage = (prop) => `The \`${prop}\` property is not available on the outer context pipeline of AdaptivePlaywrightCrawler - it is provided by the inner (static/browser) pipelines`;
|
|
201
|
+
return ContextPipeline.create().compose(async ({ request }) => ({
|
|
202
|
+
get request() {
|
|
203
|
+
return request;
|
|
204
|
+
},
|
|
205
|
+
get response() {
|
|
206
|
+
throw new Error(errorMessage('response'));
|
|
207
|
+
},
|
|
208
|
+
get page() {
|
|
209
|
+
throw new Error(errorMessage('page'));
|
|
210
|
+
},
|
|
211
|
+
get querySelector() {
|
|
212
|
+
throw new Error(errorMessage('querySelector'));
|
|
213
|
+
},
|
|
214
|
+
get querySelectorAll() {
|
|
215
|
+
throw new Error(errorMessage('querySelectorAll'));
|
|
216
|
+
},
|
|
217
|
+
get waitForSelector() {
|
|
218
|
+
throw new Error(errorMessage('waitForSelector'));
|
|
219
|
+
},
|
|
220
|
+
get parseWithCheerio() {
|
|
221
|
+
throw new Error(errorMessage('parseWithCheerio'));
|
|
222
|
+
},
|
|
223
|
+
get enqueueLinks() {
|
|
224
|
+
throw new Error(errorMessage('enqueueLinks'));
|
|
225
|
+
},
|
|
226
|
+
}));
|
|
195
227
|
}
|
|
196
228
|
async adaptCheerioContext(cheerioContext) {
|
|
197
|
-
// Capture the original response to avoid infinite recursion when the getter is copied to the context
|
|
198
|
-
const result = this.resultObjects.get(cheerioContext);
|
|
199
|
-
if (result === undefined) {
|
|
200
|
-
throw new Error('Logical error - `this.resultObjects` does not contain the result object');
|
|
201
|
-
}
|
|
202
229
|
return {
|
|
203
230
|
get page() {
|
|
204
231
|
throw new Error('Page object was used in HTTP-only request handler');
|
|
205
232
|
},
|
|
206
233
|
async querySelector(selector) {
|
|
234
|
+
return cheerioContext.$(selector).first();
|
|
235
|
+
},
|
|
236
|
+
async querySelectorAll(selector) {
|
|
207
237
|
return cheerioContext.$(selector);
|
|
208
238
|
},
|
|
209
239
|
enqueueLinks: async (options = {}) => {
|
|
210
|
-
const urls = options.
|
|
211
|
-
|
|
212
|
-
return (await this.enqueueLinks({ ...options, urls }, cheerioContext.request, result));
|
|
240
|
+
const urls = extractUrlsFromCheerio(cheerioContext.$, options.selector, options.baseUrl ?? cheerioContext.request.loadedUrl);
|
|
241
|
+
return (await this.enqueueLinks(urls, options, cheerioContext.request));
|
|
213
242
|
},
|
|
214
243
|
response: cheerioContext.response,
|
|
215
244
|
};
|
|
216
245
|
}
|
|
217
246
|
async adaptPlaywrightContext(playwrightContext) {
|
|
247
|
+
// Capture the original response to avoid infinite recursion when the getter is copied to the context
|
|
218
248
|
const originalResponse = playwrightContext.response;
|
|
219
|
-
const result = this.resultObjects.get(playwrightContext);
|
|
220
|
-
if (result === undefined) {
|
|
221
|
-
throw new Error('Logical error - `this.resultObjects` does not contain the result object');
|
|
222
|
-
}
|
|
223
249
|
return {
|
|
224
250
|
response: new Response(Uint8Array.from(await originalResponse.body()), {
|
|
225
251
|
headers: originalResponse.headers(),
|
|
@@ -227,6 +253,12 @@ export class AdaptivePlaywrightCrawler extends BasicCrawler {
|
|
|
227
253
|
statusText: originalResponse.statusText(),
|
|
228
254
|
}),
|
|
229
255
|
async querySelector(selector, timeoutMs = 5000) {
|
|
256
|
+
const locator = playwrightContext.page.locator(selector).first();
|
|
257
|
+
await locator.waitFor({ timeout: timeoutMs, state: 'attached' });
|
|
258
|
+
const $ = await playwrightContext.parseWithCheerio();
|
|
259
|
+
return $(selector).first();
|
|
260
|
+
},
|
|
261
|
+
async querySelectorAll(selector, timeoutMs = 5000) {
|
|
230
262
|
const locator = playwrightContext.page.locator(selector).first();
|
|
231
263
|
await locator.waitFor({ timeout: timeoutMs, state: 'attached' });
|
|
232
264
|
const $ = await playwrightContext.parseWithCheerio();
|
|
@@ -234,54 +266,58 @@ export class AdaptivePlaywrightCrawler extends BasicCrawler {
|
|
|
234
266
|
},
|
|
235
267
|
enqueueLinks: async (options = {}, timeoutMs = 5000) => {
|
|
236
268
|
// TODO consider using `context.parseWithCheerio` to make this universal and avoid code duplication
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
urls =
|
|
243
|
-
options.urls ??
|
|
244
|
-
(await extractUrlsFromPage(playwrightContext.page, selector, options.baseUrl ?? playwrightContext.request.loadedUrl));
|
|
245
|
-
}
|
|
246
|
-
else {
|
|
247
|
-
urls = options.urls;
|
|
248
|
-
}
|
|
249
|
-
return (await this.enqueueLinks({ ...options, urls }, playwrightContext.request, result));
|
|
269
|
+
const selector = options.selector ?? 'a';
|
|
270
|
+
const locator = playwrightContext.page.locator(selector).first();
|
|
271
|
+
await locator.waitFor({ timeout: timeoutMs, state: 'attached' });
|
|
272
|
+
const urls = await extractUrlsFromPage(playwrightContext.page, selector, options.baseUrl ?? playwrightContext.request.loadedUrl);
|
|
273
|
+
return (await this.enqueueLinks(urls, options, playwrightContext.request));
|
|
250
274
|
},
|
|
251
275
|
};
|
|
252
276
|
}
|
|
253
|
-
|
|
254
|
-
|
|
277
|
+
/**
|
|
278
|
+
* Runs one request handler attempt inside its own {@link StorageTransaction}, wrapping the inner
|
|
279
|
+
* (static or browser) context pipeline. The transaction is pushed to `transactions` *at creation
|
|
280
|
+
* time, before the `try`* - the `ok: false` branch of the returned {@link Result} carries no
|
|
281
|
+
* result, and failed attempts are routine here. The caller owns the outcome and disposal.
|
|
282
|
+
*/
|
|
283
|
+
async crawlOne(renderingType, context, useStateFunction, transactions) {
|
|
284
|
+
const transaction = createStorageTransaction({
|
|
285
|
+
policy: this.#attemptWritePolicy,
|
|
286
|
+
commitTimeoutMillis: this.internalTimeoutMillis,
|
|
287
|
+
});
|
|
288
|
+
transactions.push(transaction);
|
|
255
289
|
const logs = [];
|
|
256
290
|
const deferredCleanup = [];
|
|
257
|
-
const
|
|
258
|
-
|
|
259
|
-
pushData: result.pushData,
|
|
260
|
-
useState: this.allowStorageAccess(useStateFunction),
|
|
261
|
-
getKeyValueStore: this.allowStorageAccess(result.getKeyValueStore),
|
|
262
|
-
enqueueLinks: async (options) => {
|
|
263
|
-
return await this.enqueueLinks(options, context.request, result);
|
|
264
|
-
},
|
|
291
|
+
const attemptBoundContextHelpers = {
|
|
292
|
+
useState: useStateFunction,
|
|
265
293
|
log: this.createLogProxy(context.log, logs),
|
|
266
294
|
registerDeferredCleanup: (cleanup) => deferredCleanup.push(cleanup),
|
|
267
295
|
};
|
|
268
|
-
const subCrawlerContext = {
|
|
269
|
-
|
|
296
|
+
const subCrawlerContext = Object.defineProperties({}, Object.getOwnPropertyDescriptors(context));
|
|
297
|
+
// Mark attempt-bound helpers as non-configurable so they survive the sub-crawler context pipeline
|
|
298
|
+
// (which would otherwise override them with the sub-crawler's own versions, losing the binding).
|
|
299
|
+
for (const [key, descriptor] of Object.entries(Object.getOwnPropertyDescriptors(attemptBoundContextHelpers))) {
|
|
300
|
+
Object.defineProperty(subCrawlerContext, key, { ...descriptor, configurable: false });
|
|
301
|
+
}
|
|
270
302
|
try {
|
|
303
|
+
// Any failure - middleware or handler - ends up in the `ok: false` branch below.
|
|
304
|
+
const rethrow = (error) => {
|
|
305
|
+
throw error;
|
|
306
|
+
};
|
|
271
307
|
const callAdaptiveRequestHandler = async () => {
|
|
272
308
|
if (renderingType === 'static') {
|
|
273
|
-
await this
|
|
309
|
+
await this.#staticContextPipeline.call(subCrawlerContext, this.requestHandler.bind(this), rethrow);
|
|
274
310
|
}
|
|
275
311
|
else if (renderingType === 'clientOnly') {
|
|
276
|
-
await this
|
|
312
|
+
await this.#browserContextPipeline.call(subCrawlerContext, this.requestHandler.bind(this), rethrow);
|
|
277
313
|
}
|
|
278
314
|
};
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
return { result, ok: true, logs };
|
|
315
|
+
// this crawler overrides `runRequestHandler` and times each rendering-type run itself, so it has
|
|
316
|
+
// to resolve any per-route override too - otherwise routes would be silently ignored here
|
|
317
|
+
const routeTimeoutSecs = this.requestHandler.getTimeoutSecs?.(context.request.label);
|
|
318
|
+
const timeoutMillis = routeTimeoutSecs === undefined ? this.#individualRequestHandlerTimeoutMillis : routeTimeoutSecs * 1000;
|
|
319
|
+
await addTimeoutToPromise(async () => transaction.run(callAdaptiveRequestHandler), timeoutMillis, 'Request handler timed out');
|
|
320
|
+
return { result: transaction, ok: true, logs };
|
|
285
321
|
}
|
|
286
322
|
catch (error) {
|
|
287
323
|
return { error, ok: false, logs };
|
|
@@ -291,138 +327,245 @@ export class AdaptivePlaywrightCrawler extends BasicCrawler {
|
|
|
291
327
|
}
|
|
292
328
|
}
|
|
293
329
|
async runRequestHandler(crawlingContext) {
|
|
294
|
-
const renderingTypePrediction = this
|
|
295
|
-
const shouldDetectRenderingType = Math.random() < renderingTypePrediction.detectionProbabilityRecommendation;
|
|
330
|
+
const renderingTypePrediction = await this.#renderingTypePredictor.value.predict(crawlingContext.request);
|
|
331
|
+
const shouldDetectRenderingType = !this.#shutDown && Math.random() < renderingTypePrediction.detectionProbabilityRecommendation;
|
|
296
332
|
if (!shouldDetectRenderingType) {
|
|
297
333
|
crawlingContext.log.debug(`Predicted rendering type ${renderingTypePrediction.renderingType} for ${crawlingContext.request.url}`);
|
|
298
334
|
}
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
: plainHTTPRun.error;
|
|
314
|
-
crawlingContext.log.exception(actualError, `HTTP-only request handler failed for ${crawlingContext.request.url}`);
|
|
315
|
-
}
|
|
316
|
-
else {
|
|
317
|
-
crawlingContext.log.warning(`HTTP-only request handler returned a suspicious result for ${crawlingContext.request.url}`);
|
|
318
|
-
this.stats.trackRenderingTypeMisprediction();
|
|
319
|
-
}
|
|
320
|
-
}
|
|
321
|
-
crawlingContext.log.debug(`Running browser request handler for ${crawlingContext.request.url}`);
|
|
322
|
-
this.stats.trackBrowserRequestHandlerRun();
|
|
323
|
-
// Run the request handler in a browser. The copy of the crawler state is kept so that we can perform
|
|
324
|
-
// a rendering type detection if necessary. Without this measure, the HTTP request handler would run
|
|
325
|
-
// under different conditions, which could change its behavior. Changes done to the crawler state by
|
|
326
|
-
// the HTTP request handler will not be committed to the actual storage.
|
|
327
|
-
const stateTracker = {
|
|
328
|
-
stateCopy: null,
|
|
329
|
-
async getLiveState(defaultValue = {}) {
|
|
330
|
-
const state = await crawlingContext.useState(defaultValue);
|
|
331
|
-
if (this.stateCopy === null) {
|
|
332
|
-
this.stateCopy = JSON.parse(JSON.stringify(state));
|
|
333
|
-
}
|
|
334
|
-
return state;
|
|
335
|
-
},
|
|
336
|
-
async getStateCopy(defaultValue = {}) {
|
|
337
|
-
if (this.stateCopy === null) {
|
|
338
|
-
return defaultValue;
|
|
335
|
+
// Every transaction created for this request - up to two, since the static-then-browser
|
|
336
|
+
// fall-through and the browser-then-detection pair are mutually exclusive. Disposed in the
|
|
337
|
+
// `finally` below, not earlier: the comparators read the journals after `crawlOne` returns.
|
|
338
|
+
const transactions = [];
|
|
339
|
+
try {
|
|
340
|
+
if (renderingTypePrediction.renderingType === 'static' && !shouldDetectRenderingType) {
|
|
341
|
+
crawlingContext.log.debug(`Running HTTP-only request handler for ${crawlingContext.request.url}`);
|
|
342
|
+
this.statistics.state.httpOnlyRequestHandlerRuns++;
|
|
343
|
+
const plainHTTPRun = await this.crawlOne('static', crawlingContext, crawlingContext.useState, transactions);
|
|
344
|
+
if (plainHTTPRun.ok && this.#resultChecker(plainHTTPRun.result)) {
|
|
345
|
+
crawlingContext.log.debug(`HTTP-only request handler succeeded for ${crawlingContext.request.url}`);
|
|
346
|
+
plainHTTPRun.logs?.forEach(([log, method, ...args]) => log[method](...args));
|
|
347
|
+
await plainHTTPRun.result.commit();
|
|
348
|
+
return;
|
|
339
349
|
}
|
|
340
|
-
|
|
341
|
-
},
|
|
342
|
-
};
|
|
343
|
-
const browserRun = await this.crawlOne('clientOnly', crawlingContext, stateTracker.getLiveState.bind(stateTracker));
|
|
344
|
-
if (!browserRun.ok) {
|
|
345
|
-
throw browserRun.error;
|
|
346
|
-
}
|
|
347
|
-
await this.commitResult(crawlingContext, browserRun.result);
|
|
348
|
-
if (shouldDetectRenderingType) {
|
|
349
|
-
crawlingContext.log.debug(`Detecting rendering type for ${crawlingContext.request.url}`);
|
|
350
|
-
const plainHTTPRun = await this.crawlOne('static', crawlingContext, stateTracker.getStateCopy.bind(stateTracker));
|
|
351
|
-
const detectionResult = (() => {
|
|
350
|
+
// Execution will "fall through" and try running the request handler in a browser
|
|
352
351
|
if (!plainHTTPRun.ok) {
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
352
|
+
const actualError = plainHTTPRun.error instanceof RequestHandlerError ||
|
|
353
|
+
plainHTTPRun.error instanceof ContextPipelineInitializationError
|
|
354
|
+
? plainHTTPRun.error.cause
|
|
355
|
+
: plainHTTPRun.error;
|
|
356
|
+
if (await this.#shouldPropagateError(actualError, crawlingContext)) {
|
|
357
|
+
throw actualError;
|
|
358
|
+
}
|
|
359
|
+
crawlingContext.log.exception(actualError, `HTTP-only request handler failed for ${crawlingContext.request.url}`);
|
|
358
360
|
}
|
|
359
|
-
|
|
360
|
-
|
|
361
|
+
else {
|
|
362
|
+
crawlingContext.log.warning(`HTTP-only request handler returned a suspicious result for ${crawlingContext.request.url}`);
|
|
363
|
+
this.statistics.state.renderingTypeMispredictions++;
|
|
361
364
|
}
|
|
362
|
-
|
|
363
|
-
})
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
365
|
+
}
|
|
366
|
+
crawlingContext.log.debug(`Running browser request handler for ${crawlingContext.request.url}`);
|
|
367
|
+
this.statistics.state.browserRequestHandlerRuns++;
|
|
368
|
+
// Run the request handler in a browser. The copy of the crawler state is kept so that we can perform
|
|
369
|
+
// a rendering type detection if necessary. Without this measure, the HTTP request handler would run
|
|
370
|
+
// under different conditions, which could change its behavior. Changes done to the crawler state by
|
|
371
|
+
// the HTTP request handler will not be committed to the actual storage.
|
|
372
|
+
const stateTracker = {
|
|
373
|
+
stateCopy: null,
|
|
374
|
+
async getLiveState(defaultValue = {}) {
|
|
375
|
+
const state = await crawlingContext.useState(defaultValue);
|
|
376
|
+
if (this.stateCopy === null) {
|
|
377
|
+
this.stateCopy = JSON.parse(JSON.stringify(state));
|
|
378
|
+
}
|
|
379
|
+
return state;
|
|
380
|
+
},
|
|
381
|
+
async getStateCopy(defaultValue = {}) {
|
|
382
|
+
if (this.stateCopy === null) {
|
|
383
|
+
return defaultValue;
|
|
384
|
+
}
|
|
385
|
+
return this.stateCopy;
|
|
386
|
+
},
|
|
387
|
+
};
|
|
388
|
+
const browserRun = await this.crawlOne('clientOnly', crawlingContext, stateTracker.getLiveState.bind(stateTracker), transactions);
|
|
389
|
+
if (!browserRun.ok) {
|
|
390
|
+
throw browserRun.error;
|
|
391
|
+
}
|
|
392
|
+
browserRun.logs?.forEach(([log, method, ...args]) => log[method](...args));
|
|
393
|
+
await browserRun.result.commit();
|
|
394
|
+
if (shouldDetectRenderingType) {
|
|
395
|
+
const detectionPromise = (async () => {
|
|
396
|
+
crawlingContext.log.debug(`Detecting rendering type for ${crawlingContext.request.url}`);
|
|
397
|
+
// The detection attempt's transaction is never committed - its writes exist only for the
|
|
398
|
+
// result comparison.
|
|
399
|
+
const plainHTTPRun = await this.crawlOne('static', crawlingContext, stateTracker.getStateCopy.bind(stateTracker), transactions);
|
|
400
|
+
const detectionResult = (() => {
|
|
401
|
+
if (!plainHTTPRun.ok) {
|
|
402
|
+
return 'clientOnly';
|
|
403
|
+
}
|
|
404
|
+
const comparisonResult = this.#resultComparator(plainHTTPRun.result, browserRun.result);
|
|
405
|
+
if (comparisonResult === true || comparisonResult === 'equal') {
|
|
406
|
+
return 'static';
|
|
407
|
+
}
|
|
408
|
+
if (comparisonResult === false || comparisonResult === 'different') {
|
|
409
|
+
return 'clientOnly';
|
|
410
|
+
}
|
|
411
|
+
return undefined;
|
|
412
|
+
})();
|
|
413
|
+
crawlingContext.log.debug(`Detected rendering type ${detectionResult} for ${crawlingContext.request.url}`);
|
|
414
|
+
if (detectionResult !== undefined) {
|
|
415
|
+
// Deliberately not awaited: a predictor that persists asynchronously gets to keep
|
|
416
|
+
// batching its writes, and the drain below catches whatever is still pending.
|
|
417
|
+
const stored = this.#renderingTypePredictor.value.storeResult(crawlingContext.request, detectionResult);
|
|
418
|
+
if (stored !== undefined) {
|
|
419
|
+
// Nothing downstream awaits this, so a failed write would otherwise be silent.
|
|
420
|
+
void this.#trackDetection(Promise.resolve(stored).catch((error) => this.log.exception(error, `Failed to store the rendering type detection result for ${crawlingContext.request.url}`)));
|
|
421
|
+
}
|
|
422
|
+
}
|
|
423
|
+
})();
|
|
424
|
+
await this.#trackDetection(detectionPromise);
|
|
425
|
+
}
|
|
426
|
+
}
|
|
427
|
+
finally {
|
|
428
|
+
// A still-open transaction here belongs to a discarded attempt - roll it back, then release.
|
|
429
|
+
for (const transaction of transactions) {
|
|
430
|
+
transaction.rollback();
|
|
431
|
+
transaction.dispose();
|
|
367
432
|
}
|
|
368
433
|
}
|
|
369
434
|
}
|
|
370
|
-
async
|
|
371
|
-
await Promise.all([
|
|
372
|
-
...calls.pushData.map(async (params) => crawlingContext.pushData(...params)),
|
|
373
|
-
...calls.addRequests.map(async (params) => crawlingContext.addRequests(...params)),
|
|
374
|
-
...Object.entries(keyValueStoreChanges).map(async ([storeIdOrName, changes]) => {
|
|
375
|
-
const store = await crawlingContext.getKeyValueStore(storeIdOrName);
|
|
376
|
-
await Promise.all(Object.entries(changes).map(async ([key, { changedValue, options }]) => store.setValue(key, changedValue, options)));
|
|
377
|
-
}),
|
|
378
|
-
]);
|
|
379
|
-
}
|
|
380
|
-
allowStorageAccess(func) {
|
|
381
|
-
return async (...args) => withCheckedStorageAccess(() => { }, async () => func(...args));
|
|
382
|
-
}
|
|
383
|
-
async enqueueLinks(options, request, result) {
|
|
435
|
+
async enqueueLinks(urls, options, request) {
|
|
384
436
|
const baseUrl = resolveBaseUrlForEnqueueLinksFiltering({
|
|
385
437
|
enqueueStrategy: options?.strategy,
|
|
386
438
|
finalRequestUrl: request.loadedUrl,
|
|
387
439
|
originalRequestUrl: request.url,
|
|
388
440
|
userProvidedBaseUrl: options?.baseUrl,
|
|
389
441
|
});
|
|
390
|
-
const
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
})),
|
|
399
|
-
waitForAllRequestsToBeAdded: Promise.resolve([]),
|
|
400
|
-
};
|
|
401
|
-
};
|
|
402
|
-
// We need to use a mock request queue implementation, in order to add the requests into our result object
|
|
403
|
-
const mockRequestQueue = { addRequestsBatched };
|
|
404
|
-
return await this.enqueueLinksWithCrawlDepth({ ...options, baseUrl }, request, mockRequestQueue);
|
|
442
|
+
const requestsWithDepth = this.addCrawlDepthRequestGenerator(urls, request.crawlDepth + 1);
|
|
443
|
+
// The per-attempt transaction buffers these (the queue policy defaults to `deferred` here),
|
|
444
|
+
// so a discarded attempt's enqueues never reach the queue.
|
|
445
|
+
return await this.addRequests(requestsWithDepth, {
|
|
446
|
+
...options,
|
|
447
|
+
baseUrl,
|
|
448
|
+
strategy: options.strategy ?? EnqueueStrategy.SameHostname,
|
|
449
|
+
});
|
|
405
450
|
}
|
|
406
451
|
createLogProxy(log, logs) {
|
|
407
452
|
return new Proxy(log, {
|
|
408
|
-
get(target, propertyName
|
|
453
|
+
get(target, propertyName) {
|
|
409
454
|
if (proxyLogMethods.includes(propertyName)) {
|
|
410
455
|
return (...args) => {
|
|
411
456
|
logs.push([target, propertyName, ...args]);
|
|
412
457
|
};
|
|
413
458
|
}
|
|
414
|
-
|
|
459
|
+
const value = Reflect.get(target, propertyName, target);
|
|
460
|
+
// Bind non-intercepted methods to the target instance so private #-fields
|
|
461
|
+
// (e.g. BaseCrawleeLogger.#options, #warningsLogged) do not throw TypeError at runtime.
|
|
462
|
+
if (typeof value === 'function') {
|
|
463
|
+
return value.bind(target);
|
|
464
|
+
}
|
|
465
|
+
return value;
|
|
415
466
|
},
|
|
416
467
|
});
|
|
417
468
|
}
|
|
469
|
+
#trackDetection(promise) {
|
|
470
|
+
this.#activeDetections.add(promise);
|
|
471
|
+
// Not `finally()`: the promise it derives would reject on its own and go unhandled. A rejection here
|
|
472
|
+
// belongs to whoever awaits the original, or to `allSettled` in the drain.
|
|
473
|
+
void promise.catch(() => { }).then(() => this.#activeDetections.delete(promise));
|
|
474
|
+
return promise;
|
|
475
|
+
}
|
|
476
|
+
/**
|
|
477
|
+
* Number of rendering type detections that have not settled yet, including results the predictor is
|
|
478
|
+
* still persisting.
|
|
479
|
+
*/
|
|
480
|
+
get inFlightRenderingTypeDetectionCount() {
|
|
481
|
+
return this.#activeDetections.size;
|
|
482
|
+
}
|
|
483
|
+
/**
|
|
484
|
+
* Waits for in-flight rendering type detections to settle, bounded by `timeoutMillis` (defaults to the
|
|
485
|
+
* internal timeout).
|
|
486
|
+
*/
|
|
487
|
+
async drainRenderingDetections({ timeoutMillis } = {}) {
|
|
488
|
+
if (this.#activeDetections.size === 0) {
|
|
489
|
+
return;
|
|
490
|
+
}
|
|
491
|
+
const drained = (async () => {
|
|
492
|
+
while (this.#activeDetections.size > 0) {
|
|
493
|
+
await Promise.allSettled(Array.from(this.#activeDetections));
|
|
494
|
+
}
|
|
495
|
+
})();
|
|
496
|
+
const millis = timeoutMillis ?? this.internalTimeoutMillis;
|
|
497
|
+
// A caller opting out of the bound would otherwise get an immediate spurious timeout - `setTimeout`
|
|
498
|
+
// clamps a non-finite delay to 1ms.
|
|
499
|
+
if (!Number.isFinite(millis)) {
|
|
500
|
+
await drained;
|
|
501
|
+
return;
|
|
502
|
+
}
|
|
503
|
+
const abortTimer = new AbortController();
|
|
504
|
+
try {
|
|
505
|
+
const outcome = await Promise.race([
|
|
506
|
+
drained.then(() => 'drained'),
|
|
507
|
+
delay(millis, 'timedOut', { signal: abortTimer.signal }).catch(() => 'aborted'),
|
|
508
|
+
]);
|
|
509
|
+
if (outcome === 'timedOut') {
|
|
510
|
+
this.log.warning(`Timed out after ${millis / 1e3} seconds waiting for ${this.#activeDetections.size} rendering type detection(s) to settle - their results may be lost.`);
|
|
511
|
+
}
|
|
512
|
+
}
|
|
513
|
+
finally {
|
|
514
|
+
abortTimer.abort();
|
|
515
|
+
}
|
|
516
|
+
}
|
|
517
|
+
/**
|
|
518
|
+
* Stops the crawler immediately, but not before rendering type detections already under way (and results
|
|
519
|
+
* the predictor is still persisting) have settled - see
|
|
520
|
+
* {@link AdaptivePlaywrightCrawler.drainRenderingDetections|`drainRenderingDetections()`}. Requests
|
|
521
|
+
* that are still running are not waited for, unlike {@link BasicCrawler.stop|`stop()`}.
|
|
522
|
+
*/
|
|
418
523
|
async teardown() {
|
|
524
|
+
// Called from outside `run()` - under `keepAlive`, say - the pool keeps dispatching until
|
|
525
|
+
// `super.teardown()` aborts it, and a request starting during the drain would open a detection the
|
|
526
|
+
// drain has already passed. Closing that first makes the drain a fence.
|
|
527
|
+
this.#shutDown = true;
|
|
528
|
+
await this.drainRenderingDetections();
|
|
419
529
|
await super.teardown();
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
530
|
+
// Mirrors the owned-only `initialize()` in `init()` - without this, the predictor we built keeps its
|
|
531
|
+
// PERSIST_STATE listener registered after the crawl and never gets a final write.
|
|
532
|
+
await this.#renderingTypePredictor.ifOwned((predictor) => predictor.teardown());
|
|
533
|
+
await this.#browserCrawler.teardown();
|
|
423
534
|
}
|
|
535
|
+
async destroy() {
|
|
536
|
+
await super.destroy();
|
|
537
|
+
await this.#staticCrawler.destroy();
|
|
538
|
+
await this.#browserCrawler.destroy();
|
|
539
|
+
}
|
|
540
|
+
}
|
|
541
|
+
export function createAdaptivePlaywrightRouter(routesOrSchemas) {
|
|
542
|
+
return Router.create(routesOrSchemas);
|
|
424
543
|
}
|
|
425
|
-
|
|
426
|
-
|
|
544
|
+
/**
|
|
545
|
+
* An opt-in {@link AdaptivePlaywrightCrawlerOptions.resultComparator|`resultComparator`} that considers two
|
|
546
|
+
* request handler results equal only if *all* of their observable effects match - the pushed dataset items, the
|
|
547
|
+
* enqueued requests, and the key-value store changes. This is stricter than the default comparator, which only
|
|
548
|
+
* compares dataset items.
|
|
549
|
+
*
|
|
550
|
+
* **Beware:** enqueued URLs are compared exactly. The same page rendered in a browser and via plain HTTP often
|
|
551
|
+
* yields links that differ only in tracking query parameters, for example:
|
|
552
|
+
* - `https://sdk.apify.com/docs/guides/getting-started`
|
|
553
|
+
* - `https://sdk.apify.com/docs/guides/getting-started?__hsfp=1136113150&__hssc=7591405.1.173549427712`
|
|
554
|
+
*
|
|
555
|
+
* Such links are treated as *different*, which will make the crawler favor browser rendering for those pages.
|
|
556
|
+
*
|
|
557
|
+
* **Example usage:**
|
|
558
|
+
* ```ts
|
|
559
|
+
* const crawler = new AdaptivePlaywrightCrawler({
|
|
560
|
+
* resultComparator: fullResultComparator,
|
|
561
|
+
* async requestHandler({ pushData, enqueueLinks }) {
|
|
562
|
+
* // ...
|
|
563
|
+
* },
|
|
564
|
+
* });
|
|
565
|
+
* ```
|
|
566
|
+
*/
|
|
567
|
+
export function fullResultComparator(resultA, resultB) {
|
|
568
|
+
return (isDeepStrictEqual(resultA.datasetItems, resultB.datasetItems) &&
|
|
569
|
+
isDeepStrictEqual(resultA.enqueuedUrls, resultB.enqueuedUrls) &&
|
|
570
|
+
isDeepStrictEqual(resultA.keyValueStoreChanges, resultB.keyValueStoreChanges));
|
|
427
571
|
}
|
|
428
|
-
//# sourceMappingURL=adaptive-playwright-crawler.js.map
|