@crawlee/playwright 4.0.0-beta.2 → 4.0.0-beta.200

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. package/README.md +17 -13
  2. package/index.d.ts +4 -3
  3. package/index.js +2 -2
  4. package/internals/adaptive-playwright-crawler.d.ts +146 -91
  5. package/internals/adaptive-playwright-crawler.js +485 -268
  6. package/internals/enqueue-links/click-elements.d.ts +37 -55
  7. package/internals/enqueue-links/click-elements.js +65 -55
  8. package/internals/playwright-browser-pool.d.ts +71 -0
  9. package/internals/playwright-browser-pool.js +61 -0
  10. package/internals/playwright-crawler.d.ts +177 -172
  11. package/internals/playwright-crawler.js +103 -61
  12. package/internals/playwright-launcher.d.ts +30 -20
  13. package/internals/playwright-launcher.js +22 -17
  14. package/internals/utils/playwright-utils.d.ts +55 -50
  15. package/internals/utils/playwright-utils.js +121 -148
  16. package/internals/utils/rendering-type-prediction.d.ts +44 -13
  17. package/internals/utils/rendering-type-prediction.js +95 -29
  18. package/package.json +15 -15
  19. package/index.d.ts.map +0 -1
  20. package/index.js.map +0 -1
  21. package/internals/adaptive-playwright-crawler.d.ts.map +0 -1
  22. package/internals/adaptive-playwright-crawler.js.map +0 -1
  23. package/internals/enqueue-links/click-elements.d.ts.map +0 -1
  24. package/internals/enqueue-links/click-elements.js.map +0 -1
  25. package/internals/playwright-crawler.d.ts.map +0 -1
  26. package/internals/playwright-crawler.js.map +0 -1
  27. package/internals/playwright-launcher.d.ts.map +0 -1
  28. package/internals/playwright-launcher.js.map +0 -1
  29. package/internals/utils/playwright-utils.d.ts.map +0 -1
  30. package/internals/utils/playwright-utils.js.map +0 -1
  31. package/internals/utils/rendering-type-prediction.d.ts.map +0 -1
  32. package/internals/utils/rendering-type-prediction.js.map +0 -1
  33. package/tsconfig.build.tsbuildinfo +0 -1
@@ -1,46 +1,30 @@
1
+ import { setTimeout as delay } from 'node:timers/promises';
2
+ import { isDeepStrictEqual } from 'node:util';
3
+ import { BasicCrawler, ContextPipeline, RequestHandlerError, resolveBaseUrlForEnqueueLinksFiltering, Router, Statistics, } from '@crawlee/basic';
1
4
  import { extractUrlsFromPage } from '@crawlee/browser';
2
- import { Configuration, RequestHandlerResult, Router, Statistics, withCheckedStorageAccess } from '@crawlee/core';
3
- import { extractUrlsFromCheerio } from '@crawlee/utils';
4
- import { load } from 'cheerio';
5
- import isEqual from 'lodash.isequal';
5
+ import { CheerioCrawler } from '@crawlee/cheerio';
6
+ import { createStorageTransaction, EnqueueStrategy, OwnedOrInjected } from '@crawlee/core';
7
+ import { extractUrlsFromCheerio, parseArgument } from '@crawlee/utils/internal';
8
+ import { z } from 'zod';
6
9
  import { addTimeoutToPromise } from '@apify/timeout';
7
10
  import { PlaywrightCrawler } from './playwright-crawler.js';
8
- import { RenderingTypePredictor } from './utils/rendering-type-prediction.js';
9
- class AdaptivePlaywrightCrawlerStatistics extends Statistics {
10
- state = null; // this needs to be assigned for a valid override, but the initialization is done by a reset() call from the parent constructor
11
- constructor(options = {}) {
12
- super(options);
13
- this.reset();
14
- }
15
- reset() {
16
- super.reset();
17
- this.state.httpOnlyRequestHandlerRuns = 0;
18
- this.state.browserRequestHandlerRuns = 0;
19
- this.state.renderingTypeMispredictions = 0;
20
- }
21
- async _maybeLoadStatistics() {
22
- await super._maybeLoadStatistics();
23
- const savedState = await this.keyValueStore?.getValue(this.persistStateKey);
24
- if (!savedState) {
25
- return;
26
- }
27
- this.state.httpOnlyRequestHandlerRuns = savedState.httpOnlyRequestHandlerRuns;
28
- this.state.browserRequestHandlerRuns = savedState.browserRequestHandlerRuns;
29
- this.state.renderingTypeMispredictions = savedState.renderingTypeMispredictions;
30
- }
31
- trackHttpOnlyRequestHandlerRun() {
32
- this.state.httpOnlyRequestHandlerRuns ??= 0;
33
- this.state.httpOnlyRequestHandlerRuns += 1;
34
- }
35
- trackBrowserRequestHandlerRun() {
36
- this.state.browserRequestHandlerRuns ??= 0;
37
- this.state.browserRequestHandlerRuns += 1;
38
- }
39
- trackRenderingTypeMisprediction() {
40
- this.state.renderingTypeMispredictions ??= 0;
41
- this.state.renderingTypeMispredictions += 1;
42
- }
43
- }
11
+ import { RenderingTypePredictor, } from './utils/rendering-type-prediction.js';
12
+ const adaptiveStatisticStateSchema = z.object({
13
+ /** How many requests were handled by the HTTP-only request handler. */
14
+ httpOnlyRequestHandlerRuns: z.number().default(0),
15
+ /** How many requests were handled in a browser. */
16
+ browserRequestHandlerRuns: z.number().default(0),
17
+ /** How many times the HTTP-only handler produced a result the `resultChecker` rejected. */
18
+ renderingTypeMispredictions: z.number().default(0),
19
+ });
20
+ /**
21
+ * The {@link AdaptivePlaywrightCrawlerStatisticState} fields as a {@link Statistics} state extension, defaults
22
+ * and all. A {@link Statistics} instance to be injected into an {@link AdaptivePlaywrightCrawler} has to carry
23
+ * them - `deserialize.extend()` your own fields onto this one and pass the result as `stateExtension`.
24
+ */
25
+ export const adaptivePlaywrightCrawlerStatisticState = {
26
+ deserialize: adaptiveStatisticStateSchema,
27
+ };
44
28
  const proxyLogMethods = [
45
29
  'error',
46
30
  'exception',
@@ -80,274 +64,507 @@ const proxyLogMethods = [
80
64
  *
81
65
  * @experimental
82
66
  */
83
- export class AdaptivePlaywrightCrawler extends PlaywrightCrawler {
84
- config;
85
- adaptiveRequestHandler;
86
- renderingTypePredictor;
87
- resultChecker;
88
- resultComparator;
89
- preventDirectStorageAccess;
67
+ export class AdaptivePlaywrightCrawler extends BasicCrawler {
68
+ #renderingTypePredictor;
69
+ #resultChecker;
70
+ #shouldPropagateError;
71
+ #resultComparator;
72
+ #staticContextPipeline;
73
+ #browserContextPipeline;
74
+ #individualRequestHandlerTimeoutMillis;
75
+ /**
76
+ * The write policy of the per-attempt transactions. Defaults the request queue to `deferred`:
77
+ * a discarded attempt's enqueues must never reach the queue.
78
+ */
79
+ #attemptWritePolicy;
80
+ /** Owns the browser pool this crawler's runs use, so its per-run resources are released with ours. */
81
+ #browserCrawler;
82
+ /** Nothing of its state is per-run, but it owns a session pool that outlives one. */
83
+ #staticCrawler;
84
+ /**
85
+ * In-flight rendering type detections, plus the pending results of an asynchronous `storeResult`.
86
+ */
87
+ #activeDetections = new Set();
90
88
  /**
91
- * Default {@link Router} instance that will be used if we don't specify any {@link AdaptivePlaywrightCrawlerOptions.requestHandler|`requestHandler`}.
92
- * See {@link Router.addHandler|`router.addHandler()`} and {@link Router.addDefaultHandler|`router.addDefaultHandler()`}.
89
+ * Set once `teardown()` starts, so that requests still in the pool stop opening new detections.
93
90
  */
94
- // @ts-ignore
95
- router = Router.create();
96
- constructor(options = {}, config = Configuration.getGlobalConfig()) {
97
- const { requestHandler, renderingTypeDetectionRatio = 0.1, renderingTypePredictor, resultChecker, resultComparator, statisticsOptions, preventDirectStorageAccess = true, ...rest } = options;
98
- super(rest, config);
99
- this.config = config;
100
- this.adaptiveRequestHandler = requestHandler ?? this.router;
101
- this.renderingTypePredictor =
102
- renderingTypePredictor ?? new RenderingTypePredictor({ detectionRatio: renderingTypeDetectionRatio });
103
- this.resultChecker = resultChecker ?? (() => true);
91
+ #shutDown = false;
92
+ constructor(options = {}) {
93
+ const { requestHandler, renderingTypeDetectionRatio = 0.1, renderingTypePredictor, resultChecker, shouldPropagateError, resultComparator, statistics, requestHandlerTimeoutSecs = 60, errorHandler, failedRequestHandler, preNavigationHooks = [], postNavigationHooks = [], extendContext, transactionalStorage, launchContext, headless, browserPool, remoteBrowser, ...rest } = options;
94
+ // The user's value is replaced by `false` in the `super` call below — validate it separately,
95
+ // wrapped in an object so the error still names the field.
96
+ parseArgument({ transactionalStorage }, z.object({ transactionalStorage: BasicCrawler.optionsShape.transactionalStorage }), 'AdaptivePlaywrightCrawlerOptions');
97
+ // The extra fields are only tracked if the injected instance was built with them - the types enforce that,
98
+ // but plain JS callers would otherwise silently increment `undefined` into a sticky `NaN`. Extend
99
+ // `adaptivePlaywrightCrawlerStatisticState` to satisfy this.
100
+ if (statistics !== undefined) {
101
+ parseArgument(statistics.state, z.object({
102
+ httpOnlyRequestHandlerRuns: z.number(),
103
+ browserRequestHandlerRuns: z.number(),
104
+ renderingTypeMispredictions: z.number(),
105
+ }), 'statistics.state');
106
+ }
107
+ // Per-attempt buffering is load-bearing here: the handler runs up to twice per request and the
108
+ // losing attempt's writes must be discardable.
109
+ if (transactionalStorage === false) {
110
+ throw new Error('AdaptivePlaywrightCrawler requires transactional storage - it runs the request handler ' +
111
+ 'multiple times per request and must be able to discard the storage writes of losing ' +
112
+ 'attempts. `transactionalStorage: false` is therefore not supported; a write policy ' +
113
+ 'object is accepted and forwarded to the per-attempt transactions.');
114
+ }
115
+ super({
116
+ ...rest,
117
+ errorHandler,
118
+ failedRequestHandler,
119
+ requestHandler,
120
+ requestHandlerTimeoutSecs,
121
+ // The base would build a `Statistics` without the adaptive fields, so provide a default that has them.
122
+ // The cast covers a `StatisticStateExtension` that adds further fields - those can only come from an
123
+ // injected instance, in which case this default is never built.
124
+ statistics: statistics ??
125
+ new Statistics({
126
+ logMessage: `${AdaptivePlaywrightCrawler.name} request statistics:`,
127
+ stateExtension: adaptivePlaywrightCrawlerStatisticState,
128
+ }),
129
+ contextPipelineBuilder: () => this.#buildContextPipeline(),
130
+ // The base crawler must not wrap requests in a transaction of its own - this crawler opens
131
+ // one per request handler attempt in `crawlOne` instead, forwarding the write policy of the
132
+ // user-facing option (validated above) to those.
133
+ transactionalStorage: false,
134
+ });
135
+ this.#individualRequestHandlerTimeoutMillis = requestHandlerTimeoutSecs * 1000;
136
+ // `renderingTypeDetectionRatio` only configures the default predictor - an injected one brings its own
137
+ // detection ratio (and its own state), so the option is ignored in that case.
138
+ this.#renderingTypePredictor = OwnedOrInjected.resolve(renderingTypePredictor, () => new RenderingTypePredictor({ detectionRatio: renderingTypeDetectionRatio }));
139
+ this.#attemptWritePolicy = {
140
+ requestQueue: 'deferred',
141
+ ...(typeof transactionalStorage === 'object' ? transactionalStorage : {}),
142
+ };
143
+ this.#resultChecker = resultChecker ?? (() => true);
144
+ this.#shouldPropagateError = shouldPropagateError ?? (() => false);
104
145
  if (resultComparator !== undefined) {
105
- this.resultComparator = resultComparator;
146
+ this.#resultComparator = resultComparator;
106
147
  }
107
148
  else if (resultChecker !== undefined) {
108
- this.resultComparator = (resultA, resultB) => this.resultChecker(resultA) && this.resultChecker(resultB);
149
+ this.#resultComparator = (resultA, resultB) => this.#resultChecker(resultA) && this.#resultChecker(resultB);
109
150
  }
110
151
  else {
111
- this.resultComparator = (resultA, resultB) => {
152
+ this.#resultComparator = (resultA, resultB) => {
112
153
  return (resultA.datasetItems.length === resultB.datasetItems.length &&
113
154
  resultA.datasetItems.every((itemA, i) => {
114
155
  const itemB = resultB.datasetItems[i];
115
- return isEqual(itemA, itemB);
156
+ return isDeepStrictEqual(itemA, itemB);
116
157
  }));
117
158
  };
118
159
  }
119
- this.stats = new AdaptivePlaywrightCrawlerStatistics({
120
- logMessage: `${this.log.getOptions().prefix} request statistics:`,
121
- config,
122
- ...statisticsOptions,
160
+ // `extendContext` is forwarded to the inner crawlers, which run it *before* navigation (see
161
+ // `BasicCrawler`), keeping the behavior consistent with the non-adaptive crawlers: the
162
+ // extension is visible to the pre/post-navigation hooks and the request handler, but cannot
163
+ // access navigation-dependent members (`page`, `response`, `$`, ...).
164
+ //
165
+ // The adaptive hooks target a subset context (`AdaptiveHookContext`); the casts to the inner
166
+ // crawlers' `PlaywrightHook` type relax that nominal difference. The `ContextPipeline` merges
167
+ // each hook's overrides at runtime regardless of the static type.
168
+ const staticCrawler = new CheerioCrawler({
169
+ ...rest,
170
+ statistics: new Statistics({ persistenceOptions: { enable: false } }),
171
+ preNavigationHooks,
172
+ postNavigationHooks,
173
+ extendContext,
174
+ });
175
+ const browserCrawler = new PlaywrightCrawler({
176
+ ...rest,
177
+ statistics: new Statistics({ persistenceOptions: { enable: false } }),
178
+ preNavigationHooks: preNavigationHooks,
179
+ postNavigationHooks: postNavigationHooks,
180
+ extendContext,
181
+ launchContext,
182
+ headless,
183
+ browserPool,
184
+ remoteBrowser,
123
185
  });
124
- this.preventDirectStorageAccess = preventDirectStorageAccess;
186
+ this.#staticCrawler = staticCrawler;
187
+ this.#browserCrawler = browserCrawler;
188
+ this.#staticContextPipeline = staticCrawler.contextPipeline.compose(this.adaptCheerioContext.bind(this));
189
+ this.#browserContextPipeline = browserCrawler.contextPipeline.compose(this.adaptPlaywrightContext.bind(this));
125
190
  }
126
- async _runRequestHandler(crawlingContext) {
127
- const renderingTypePrediction = this.renderingTypePredictor.predict(crawlingContext.request);
128
- const shouldDetectRenderingType = Math.random() < renderingTypePrediction.detectionProbabilityRecommendation;
129
- if (!shouldDetectRenderingType) {
130
- crawlingContext.log.debug(`Predicted rendering type ${renderingTypePrediction.renderingType} for ${crawlingContext.request.url}`);
131
- }
132
- if (renderingTypePrediction.renderingType === 'static' && !shouldDetectRenderingType) {
133
- crawlingContext.log.debug(`Running HTTP-only request handler for ${crawlingContext.request.url}`);
134
- this.stats.trackHttpOnlyRequestHandlerRun();
135
- const plainHTTPRun = await this.runRequestHandlerWithPlainHTTP(crawlingContext);
136
- if (plainHTTPRun.ok && this.resultChecker(plainHTTPRun.result)) {
137
- crawlingContext.log.debug(`HTTP-only request handler succeeded for ${crawlingContext.request.url}`);
138
- plainHTTPRun.logs?.forEach(([log, method, ...args]) => log[method](...args));
139
- await this.commitResult(crawlingContext, plainHTTPRun.result);
140
- return;
141
- }
142
- if (!plainHTTPRun.ok) {
143
- crawlingContext.log.exception(plainHTTPRun.error, `HTTP-only request handler failed for ${crawlingContext.request.url}`);
144
- }
145
- else {
146
- crawlingContext.log.warning(`HTTP-only request handler returned a suspicious result for ${crawlingContext.request.url}`);
147
- this.stats.trackRenderingTypeMisprediction();
148
- }
149
- }
150
- crawlingContext.log.debug(`Running browser request handler for ${crawlingContext.request.url}`);
151
- this.stats.trackBrowserRequestHandlerRun();
152
- // Run the request handler in a browser. The copy of the crawler state is kept so that we can perform
153
- // a rendering type detection if necessary. Without this measure, the HTTP request handler would run
154
- // under different conditions, which could change its behavior. Changes done to the crawler state by
155
- // the HTTP request handler will not be committed to the actual storage.
156
- const { result: browserRun, initialStateCopy } = await this.runRequestHandlerInBrowser(crawlingContext);
157
- if (!browserRun.ok) {
158
- throw browserRun.error;
159
- }
160
- await this.commitResult(crawlingContext, browserRun.result);
161
- if (shouldDetectRenderingType) {
162
- crawlingContext.log.debug(`Detecting rendering type for ${crawlingContext.request.url}`);
163
- const plainHTTPRun = await this.runRequestHandlerWithPlainHTTP(crawlingContext, initialStateCopy);
164
- const detectionResult = (() => {
165
- if (!plainHTTPRun.ok) {
166
- return 'clientOnly';
167
- }
168
- if (this.resultComparator(plainHTTPRun.result, browserRun.result)) {
169
- return 'static';
170
- }
171
- return 'clientOnly';
172
- })();
173
- crawlingContext.log.debug(`Detected rendering type ${detectionResult} for ${crawlingContext.request.url}`);
174
- this.renderingTypePredictor.storeResult(crawlingContext.request, detectionResult);
175
- }
191
+ async init() {
192
+ // A crawler can be run again after a teardown.
193
+ this.#shutDown = false;
194
+ // Only the predictor we built ourselves is ours to initialize - an injected one is borrowed, so its
195
+ // lifecycle (including restoring persisted state) stays with whoever created it.
196
+ await this.#renderingTypePredictor.ifOwned((predictor) => predictor.initialize());
197
+ return await super.init();
176
198
  }
177
- async commitResult(crawlingContext, { calls, keyValueStoreChanges }) {
178
- await Promise.all([
179
- ...calls.pushData.map(async (params) => crawlingContext.pushData(...params)),
180
- ...calls.enqueueLinks.map(async (params) => await crawlingContext.enqueueLinks(...params)),
181
- ...calls.addRequests.map(async (params) => crawlingContext.addRequests(...params)),
182
- ...Object.entries(keyValueStoreChanges).map(async ([storeIdOrName, changes]) => {
183
- const store = await crawlingContext.getKeyValueStore(storeIdOrName);
184
- await Promise.all(Object.entries(changes).map(async ([key, { changedValue, options }]) => store.setValue(key, changedValue, options)));
185
- }),
186
- ]);
199
+ #buildContextPipeline() {
200
+ const errorMessage = (prop) => `The \`${prop}\` property is not available on the outer context pipeline of AdaptivePlaywrightCrawler - it is provided by the inner (static/browser) pipelines`;
201
+ return ContextPipeline.create().compose(async ({ request }) => ({
202
+ get request() {
203
+ return request;
204
+ },
205
+ get response() {
206
+ throw new Error(errorMessage('response'));
207
+ },
208
+ get page() {
209
+ throw new Error(errorMessage('page'));
210
+ },
211
+ get querySelector() {
212
+ throw new Error(errorMessage('querySelector'));
213
+ },
214
+ get querySelectorAll() {
215
+ throw new Error(errorMessage('querySelectorAll'));
216
+ },
217
+ get waitForSelector() {
218
+ throw new Error(errorMessage('waitForSelector'));
219
+ },
220
+ get parseWithCheerio() {
221
+ throw new Error(errorMessage('parseWithCheerio'));
222
+ },
223
+ get enqueueLinks() {
224
+ throw new Error(errorMessage('enqueueLinks'));
225
+ },
226
+ }));
187
227
  }
188
- allowStorageAccess(func) {
189
- return async (...args) => withCheckedStorageAccess(() => { }, async () => func(...args));
228
+ async adaptCheerioContext(cheerioContext) {
229
+ return {
230
+ get page() {
231
+ throw new Error('Page object was used in HTTP-only request handler');
232
+ },
233
+ async querySelector(selector) {
234
+ return cheerioContext.$(selector).first();
235
+ },
236
+ async querySelectorAll(selector) {
237
+ return cheerioContext.$(selector);
238
+ },
239
+ enqueueLinks: async (options = {}) => {
240
+ const urls = extractUrlsFromCheerio(cheerioContext.$, options.selector, options.baseUrl ?? cheerioContext.request.loadedUrl);
241
+ return (await this.enqueueLinks(urls, options, cheerioContext.request));
242
+ },
243
+ response: cheerioContext.response,
244
+ };
190
245
  }
191
- async runRequestHandlerInBrowser(crawlingContext) {
192
- const result = new RequestHandlerResult(this.config, AdaptivePlaywrightCrawler.CRAWLEE_STATE_KEY);
193
- let initialStateCopy;
246
+ async adaptPlaywrightContext(playwrightContext) {
247
+ // Capture the original response to avoid infinite recursion when the getter is copied to the context
248
+ const originalResponse = playwrightContext.response;
249
+ return {
250
+ response: new Response(Uint8Array.from(await originalResponse.body()), {
251
+ headers: originalResponse.headers(),
252
+ status: originalResponse.status(),
253
+ statusText: originalResponse.statusText(),
254
+ }),
255
+ async querySelector(selector, timeoutMs = 5000) {
256
+ const locator = playwrightContext.page.locator(selector).first();
257
+ await locator.waitFor({ timeout: timeoutMs, state: 'attached' });
258
+ const $ = await playwrightContext.parseWithCheerio();
259
+ return $(selector).first();
260
+ },
261
+ async querySelectorAll(selector, timeoutMs = 5000) {
262
+ const locator = playwrightContext.page.locator(selector).first();
263
+ await locator.waitFor({ timeout: timeoutMs, state: 'attached' });
264
+ const $ = await playwrightContext.parseWithCheerio();
265
+ return $(selector);
266
+ },
267
+ enqueueLinks: async (options = {}, timeoutMs = 5000) => {
268
+ // TODO consider using `context.parseWithCheerio` to make this universal and avoid code duplication
269
+ const selector = options.selector ?? 'a';
270
+ const locator = playwrightContext.page.locator(selector).first();
271
+ await locator.waitFor({ timeout: timeoutMs, state: 'attached' });
272
+ const urls = await extractUrlsFromPage(playwrightContext.page, selector, options.baseUrl ?? playwrightContext.request.loadedUrl);
273
+ return (await this.enqueueLinks(urls, options, playwrightContext.request));
274
+ },
275
+ };
276
+ }
277
+ /**
278
+ * Runs one request handler attempt inside its own {@link StorageTransaction}, wrapping the inner
279
+ * (static or browser) context pipeline. The transaction is pushed to `transactions` *at creation
280
+ * time, before the `try`* - the `ok: false` branch of the returned {@link Result} carries no
281
+ * result, and failed attempts are routine here. The caller owns the outcome and disposal.
282
+ */
283
+ async crawlOne(renderingType, context, useStateFunction, transactions) {
284
+ const transaction = createStorageTransaction({
285
+ policy: this.#attemptWritePolicy,
286
+ commitTimeoutMillis: this.internalTimeoutMillis,
287
+ });
288
+ transactions.push(transaction);
289
+ const logs = [];
290
+ const deferredCleanup = [];
291
+ const attemptBoundContextHelpers = {
292
+ useState: useStateFunction,
293
+ log: this.createLogProxy(context.log, logs),
294
+ registerDeferredCleanup: (cleanup) => deferredCleanup.push(cleanup),
295
+ };
296
+ const subCrawlerContext = Object.defineProperties({}, Object.getOwnPropertyDescriptors(context));
297
+ // Mark attempt-bound helpers as non-configurable so they survive the sub-crawler context pipeline
298
+ // (which would otherwise override them with the sub-crawler's own versions, losing the binding).
299
+ for (const [key, descriptor] of Object.entries(Object.getOwnPropertyDescriptors(attemptBoundContextHelpers))) {
300
+ Object.defineProperty(subCrawlerContext, key, { ...descriptor, configurable: false });
301
+ }
194
302
  try {
195
- await super._runRequestHandler.call(new Proxy(this, {
196
- get: (target, propertyName, receiver) => {
197
- if (propertyName === 'userProvidedRequestHandler') {
198
- return async (playwrightContext) => withCheckedStorageAccess(() => {
199
- if (this.preventDirectStorageAccess) {
200
- throw new Error('Directly accessing storage in a request handler is not allowed in AdaptivePlaywrightCrawler');
201
- }
202
- }, () => this.adaptiveRequestHandler({
203
- id: crawlingContext.id,
204
- session: crawlingContext.session,
205
- proxyInfo: crawlingContext.proxyInfo,
206
- request: crawlingContext.request,
207
- response: {
208
- url: crawlingContext.response.url(),
209
- statusCode: crawlingContext.response.status(),
210
- headers: crawlingContext.response.headers(),
211
- trailers: {},
212
- complete: true,
213
- redirectUrls: [],
214
- },
215
- log: crawlingContext.log,
216
- page: crawlingContext.page,
217
- querySelector: async (selector, timeoutMs = 5_000) => {
218
- const locator = playwrightContext.page.locator(selector).first();
219
- await locator.waitFor({ timeout: timeoutMs, state: 'attached' });
220
- const $ = await playwrightContext.parseWithCheerio();
221
- return $(selector);
222
- },
223
- async waitForSelector(selector, timeoutMs = 5_000) {
224
- const locator = playwrightContext.page.locator(selector).first();
225
- await locator.waitFor({ timeout: timeoutMs, state: 'attached' });
226
- },
227
- async parseWithCheerio(selector, timeoutMs = 5_000) {
228
- if (selector) {
229
- const locator = playwrightContext.page.locator(selector).first();
230
- await locator.waitFor({ timeout: timeoutMs, state: 'attached' });
231
- }
232
- return playwrightContext.parseWithCheerio();
233
- },
234
- async enqueueLinks(options = {}, timeoutMs = 5_000) {
235
- const selector = options.selector ?? 'a';
236
- const locator = playwrightContext.page.locator(selector).first();
237
- await locator.waitFor({ timeout: timeoutMs, state: 'attached' });
238
- const urls = await extractUrlsFromPage(playwrightContext.page, selector, options.baseUrl ??
239
- playwrightContext.request.loadedUrl ??
240
- playwrightContext.request.url);
241
- await result.enqueueLinks({ ...options, urls });
242
- },
243
- addRequests: result.addRequests,
244
- pushData: result.pushData,
245
- useState: this.allowStorageAccess(async (defaultValue) => {
246
- const state = await result.useState(defaultValue);
247
- if (initialStateCopy === undefined) {
248
- initialStateCopy = JSON.parse(JSON.stringify(state));
249
- }
250
- return state;
251
- }),
252
- getKeyValueStore: this.allowStorageAccess(result.getKeyValueStore),
253
- }));
254
- }
255
- return Reflect.get(target, propertyName, receiver);
256
- },
257
- }), crawlingContext);
258
- return { result: { result, ok: true }, initialStateCopy };
303
+ // Any failure - middleware or handler - ends up in the `ok: false` branch below.
304
+ const rethrow = (error) => {
305
+ throw error;
306
+ };
307
+ const callAdaptiveRequestHandler = async () => {
308
+ if (renderingType === 'static') {
309
+ await this.#staticContextPipeline.call(subCrawlerContext, this.requestHandler.bind(this), rethrow);
310
+ }
311
+ else if (renderingType === 'clientOnly') {
312
+ await this.#browserContextPipeline.call(subCrawlerContext, this.requestHandler.bind(this), rethrow);
313
+ }
314
+ };
315
+ // this crawler overrides `runRequestHandler` and times each rendering-type run itself, so it has
316
+ // to resolve any per-route override too - otherwise routes would be silently ignored here
317
+ const routeTimeoutSecs = this.requestHandler.getTimeoutSecs?.(context.request.label);
318
+ const timeoutMillis = routeTimeoutSecs === undefined ? this.#individualRequestHandlerTimeoutMillis : routeTimeoutSecs * 1000;
319
+ await addTimeoutToPromise(async () => transaction.run(callAdaptiveRequestHandler), timeoutMillis, 'Request handler timed out');
320
+ return { result: transaction, ok: true, logs };
259
321
  }
260
322
  catch (error) {
261
- return { result: { error, ok: false }, initialStateCopy };
323
+ return { error, ok: false, logs };
324
+ }
325
+ finally {
326
+ await Promise.all(deferredCleanup.map((cleanup) => cleanup()));
262
327
  }
263
328
  }
264
- async runRequestHandlerWithPlainHTTP(crawlingContext, oldStateCopy) {
265
- const result = new RequestHandlerResult(this.config, AdaptivePlaywrightCrawler.CRAWLEE_STATE_KEY);
266
- const logs = [];
267
- const pageGotoOptions = { timeout: this.navigationTimeoutMillis }; // Irrelevant, but required by BrowserCrawler
329
+ async runRequestHandler(crawlingContext) {
330
+ const renderingTypePrediction = await this.#renderingTypePredictor.value.predict(crawlingContext.request);
331
+ const shouldDetectRenderingType = !this.#shutDown && Math.random() < renderingTypePrediction.detectionProbabilityRecommendation;
332
+ if (!shouldDetectRenderingType) {
333
+ crawlingContext.log.debug(`Predicted rendering type ${renderingTypePrediction.renderingType} for ${crawlingContext.request.url}`);
334
+ }
335
+ // Every transaction created for this request - up to two, since the static-then-browser
336
+ // fall-through and the browser-then-detection pair are mutually exclusive. Disposed in the
337
+ // `finally` below, not earlier: the comparators read the journals after `crawlOne` returns.
338
+ const transactions = [];
268
339
  try {
269
- await withCheckedStorageAccess(() => {
270
- if (this.preventDirectStorageAccess) {
271
- throw new Error('Directly accessing storage in a request handler is not allowed in AdaptivePlaywrightCrawler');
340
+ if (renderingTypePrediction.renderingType === 'static' && !shouldDetectRenderingType) {
341
+ crawlingContext.log.debug(`Running HTTP-only request handler for ${crawlingContext.request.url}`);
342
+ this.statistics.state.httpOnlyRequestHandlerRuns++;
343
+ const plainHTTPRun = await this.crawlOne('static', crawlingContext, crawlingContext.useState, transactions);
344
+ if (plainHTTPRun.ok && this.#resultChecker(plainHTTPRun.result)) {
345
+ crawlingContext.log.debug(`HTTP-only request handler succeeded for ${crawlingContext.request.url}`);
346
+ plainHTTPRun.logs?.forEach(([log, method, ...args]) => log[method](...args));
347
+ await plainHTTPRun.result.commit();
348
+ return;
272
349
  }
273
- }, async () => addTimeoutToPromise(async () => {
274
- const hookContext = {
275
- id: crawlingContext.id,
276
- session: crawlingContext.session,
277
- proxyInfo: crawlingContext.proxyInfo,
278
- request: crawlingContext.request,
279
- log: this.createLogProxy(crawlingContext.log, logs),
280
- };
281
- await this._executeHooks(this.preNavigationHooks, {
282
- ...hookContext,
283
- get page() {
284
- throw new Error('Page object was used in HTTP-only pre-navigation hook');
285
- },
286
- }, // This is safe because `executeHooks` just passes the context to the hooks which accept the partial context
287
- pageGotoOptions);
288
- const response = await crawlingContext.sendRequest({});
289
- const loadedUrl = response.url;
290
- crawlingContext.request.loadedUrl = loadedUrl;
291
- const $ = load(response.body);
292
- await this.adaptiveRequestHandler({
293
- ...hookContext,
294
- request: crawlingContext.request,
295
- response,
296
- get page() {
297
- throw new Error('Page object was used in HTTP-only request handler');
298
- },
299
- async querySelector(selector, _timeoutMs) {
300
- return $(selector);
301
- },
302
- async waitForSelector(selector, _timeoutMs) {
303
- if ($(selector).get().length === 0) {
304
- throw new Error(`Selector '${selector}' not found.`);
350
+ // Execution will "fall through" and try running the request handler in a browser
351
+ if (!plainHTTPRun.ok) {
352
+ const actualError = plainHTTPRun.error instanceof RequestHandlerError
353
+ ? plainHTTPRun.error.cause
354
+ : plainHTTPRun.error;
355
+ if (await this.#shouldPropagateError(actualError, crawlingContext)) {
356
+ throw actualError;
357
+ }
358
+ crawlingContext.log.exception(actualError, `HTTP-only request handler failed for ${crawlingContext.request.url}`);
359
+ }
360
+ else {
361
+ crawlingContext.log.warning(`HTTP-only request handler returned a suspicious result for ${crawlingContext.request.url}`);
362
+ this.statistics.state.renderingTypeMispredictions++;
363
+ }
364
+ }
365
+ crawlingContext.log.debug(`Running browser request handler for ${crawlingContext.request.url}`);
366
+ this.statistics.state.browserRequestHandlerRuns++;
367
+ // Run the request handler in a browser. The copy of the crawler state is kept so that we can perform
368
+ // a rendering type detection if necessary. Without this measure, the HTTP request handler would run
369
+ // under different conditions, which could change its behavior. Changes done to the crawler state by
370
+ // the HTTP request handler will not be committed to the actual storage.
371
+ const stateTracker = {
372
+ stateCopy: null,
373
+ async getLiveState(defaultValue = {}) {
374
+ const state = await crawlingContext.useState(defaultValue);
375
+ if (this.stateCopy === null) {
376
+ this.stateCopy = JSON.parse(JSON.stringify(state));
377
+ }
378
+ return state;
379
+ },
380
+ async getStateCopy(defaultValue = {}) {
381
+ if (this.stateCopy === null) {
382
+ return defaultValue;
383
+ }
384
+ return this.stateCopy;
385
+ },
386
+ };
387
+ const browserRun = await this.crawlOne('clientOnly', crawlingContext, stateTracker.getLiveState.bind(stateTracker), transactions);
388
+ if (!browserRun.ok) {
389
+ throw browserRun.error;
390
+ }
391
+ browserRun.logs?.forEach(([log, method, ...args]) => log[method](...args));
392
+ await browserRun.result.commit();
393
+ if (shouldDetectRenderingType) {
394
+ const detectionPromise = (async () => {
395
+ crawlingContext.log.debug(`Detecting rendering type for ${crawlingContext.request.url}`);
396
+ // The detection attempt's transaction is never committed - its writes exist only for the
397
+ // result comparison.
398
+ const plainHTTPRun = await this.crawlOne('static', crawlingContext, stateTracker.getStateCopy.bind(stateTracker), transactions);
399
+ const detectionResult = (() => {
400
+ if (!plainHTTPRun.ok) {
401
+ return 'clientOnly';
305
402
  }
306
- },
307
- async parseWithCheerio(selector, _timeoutMs) {
308
- if (selector && $(selector).get().length === 0) {
309
- throw new Error(`Selector '${selector}' not found.`);
403
+ const comparisonResult = this.#resultComparator(plainHTTPRun.result, browserRun.result);
404
+ if (comparisonResult === true || comparisonResult === 'equal') {
405
+ return 'static';
310
406
  }
311
- return $;
312
- },
313
- async enqueueLinks(options = {}) {
314
- const urls = extractUrlsFromCheerio($, options.selector, options.baseUrl ?? loadedUrl);
315
- await result.enqueueLinks({ ...options, urls });
316
- },
317
- addRequests: result.addRequests,
318
- pushData: result.pushData,
319
- useState: async (defaultValue) => {
320
- // return the old state before the browser handler was executed
321
- // when rerunning the handler via HTTP for detection
322
- if (oldStateCopy !== undefined) {
323
- return oldStateCopy ?? defaultValue; // fallback to the default for `null`
407
+ if (comparisonResult === false || comparisonResult === 'different') {
408
+ return 'clientOnly';
324
409
  }
325
- return this.allowStorageAccess(result.useState)(defaultValue);
326
- },
327
- getKeyValueStore: this.allowStorageAccess(result.getKeyValueStore),
328
- });
329
- await this._executeHooks(this.postNavigationHooks, crawlingContext, pageGotoOptions);
330
- }, this.requestHandlerTimeoutInnerMillis, 'Request handler timed out'));
331
- return { result, logs, ok: true };
410
+ return undefined;
411
+ })();
412
+ crawlingContext.log.debug(`Detected rendering type ${detectionResult} for ${crawlingContext.request.url}`);
413
+ if (detectionResult !== undefined) {
414
+ // Deliberately not awaited: a predictor that persists asynchronously gets to keep
415
+ // batching its writes, and the drain below catches whatever is still pending.
416
+ const stored = this.#renderingTypePredictor.value.storeResult(crawlingContext.request, detectionResult);
417
+ if (stored !== undefined) {
418
+ // Nothing downstream awaits this, so a failed write would otherwise be silent.
419
+ void this.#trackDetection(Promise.resolve(stored).catch((error) => this.log.exception(error, `Failed to store the rendering type detection result for ${crawlingContext.request.url}`)));
420
+ }
421
+ }
422
+ })();
423
+ await this.#trackDetection(detectionPromise);
424
+ }
332
425
  }
333
- catch (error) {
334
- return { error, logs, ok: false };
426
+ finally {
427
+ // A still-open transaction here belongs to a discarded attempt - roll it back, then release.
428
+ for (const transaction of transactions) {
429
+ transaction.rollback();
430
+ transaction.dispose();
431
+ }
335
432
  }
336
433
  }
434
+ async enqueueLinks(urls, options, request) {
435
+ const baseUrl = resolveBaseUrlForEnqueueLinksFiltering({
436
+ enqueueStrategy: options?.strategy,
437
+ finalRequestUrl: request.loadedUrl,
438
+ originalRequestUrl: request.url,
439
+ userProvidedBaseUrl: options?.baseUrl,
440
+ });
441
+ const requestsWithDepth = this.addCrawlDepthRequestGenerator(urls, request.crawlDepth + 1);
442
+ // The per-attempt transaction buffers these (the queue policy defaults to `deferred` here),
443
+ // so a discarded attempt's enqueues never reach the queue.
444
+ return await this.addRequests(requestsWithDepth, {
445
+ ...options,
446
+ baseUrl,
447
+ strategy: options.strategy ?? EnqueueStrategy.SameHostname,
448
+ });
449
+ }
337
450
  createLogProxy(log, logs) {
338
451
  return new Proxy(log, {
339
- get(target, propertyName, receiver) {
452
+ get(target, propertyName) {
340
453
  if (proxyLogMethods.includes(propertyName)) {
341
454
  return (...args) => {
342
455
  logs.push([target, propertyName, ...args]);
343
456
  };
344
457
  }
345
- return Reflect.get(target, propertyName, receiver);
458
+ const value = Reflect.get(target, propertyName, target);
459
+ // Bind non-intercepted methods to the target instance so private #-fields
460
+ // (e.g. BaseCrawleeLogger.#options, #warningsLogged) do not throw TypeError at runtime.
461
+ if (typeof value === 'function') {
462
+ return value.bind(target);
463
+ }
464
+ return value;
346
465
  },
347
466
  });
348
467
  }
468
+ #trackDetection(promise) {
469
+ this.#activeDetections.add(promise);
470
+ // Not `finally()`: the promise it derives would reject on its own and go unhandled. A rejection here
471
+ // belongs to whoever awaits the original, or to `allSettled` in the drain.
472
+ void promise.catch(() => { }).then(() => this.#activeDetections.delete(promise));
473
+ return promise;
474
+ }
475
+ /**
476
+ * Number of rendering type detections that have not settled yet, including results the predictor is
477
+ * still persisting.
478
+ */
479
+ get inFlightRenderingTypeDetectionCount() {
480
+ return this.#activeDetections.size;
481
+ }
482
+ /**
483
+ * Waits for in-flight rendering type detections to settle, bounded by `timeoutMillis` (defaults to the
484
+ * internal timeout).
485
+ */
486
+ async drainRenderingDetections({ timeoutMillis } = {}) {
487
+ if (this.#activeDetections.size === 0) {
488
+ return;
489
+ }
490
+ const drained = (async () => {
491
+ while (this.#activeDetections.size > 0) {
492
+ await Promise.allSettled(Array.from(this.#activeDetections));
493
+ }
494
+ })();
495
+ const millis = timeoutMillis ?? this.internalTimeoutMillis;
496
+ // A caller opting out of the bound would otherwise get an immediate spurious timeout - `setTimeout`
497
+ // clamps a non-finite delay to 1ms.
498
+ if (!Number.isFinite(millis)) {
499
+ await drained;
500
+ return;
501
+ }
502
+ const abortTimer = new AbortController();
503
+ try {
504
+ const outcome = await Promise.race([
505
+ drained.then(() => 'drained'),
506
+ delay(millis, 'timedOut', { signal: abortTimer.signal }).catch(() => 'aborted'),
507
+ ]);
508
+ if (outcome === 'timedOut') {
509
+ this.log.warning(`Timed out after ${millis / 1e3} seconds waiting for ${this.#activeDetections.size} rendering type detection(s) to settle - their results may be lost.`);
510
+ }
511
+ }
512
+ finally {
513
+ abortTimer.abort();
514
+ }
515
+ }
516
+ /**
517
+ * Stops the crawler immediately, but not before rendering type detections already under way (and results
518
+ * the predictor is still persisting) have settled - see
519
+ * {@link AdaptivePlaywrightCrawler.drainRenderingDetections|`drainRenderingDetections()`}. Requests
520
+ * that are still running are not waited for, unlike {@link BasicCrawler.stop|`stop()`}.
521
+ */
522
+ async teardown() {
523
+ // Called from outside `run()` - under `keepAlive`, say - the pool keeps dispatching until
524
+ // `super.teardown()` aborts it, and a request starting during the drain would open a detection the
525
+ // drain has already passed. Closing that first makes the drain a fence.
526
+ this.#shutDown = true;
527
+ await this.drainRenderingDetections();
528
+ await super.teardown();
529
+ // Mirrors the owned-only `initialize()` in `init()` - without this, the predictor we built keeps its
530
+ // PERSIST_STATE listener registered after the crawl and never gets a final write.
531
+ await this.#renderingTypePredictor.ifOwned((predictor) => predictor.teardown());
532
+ await this.#browserCrawler.teardown();
533
+ }
534
+ async destroy() {
535
+ await super.destroy();
536
+ await this.#staticCrawler.destroy();
537
+ await this.#browserCrawler.destroy();
538
+ }
539
+ }
540
+ export function createAdaptivePlaywrightRouter(routesOrSchemas) {
541
+ return Router.create(routesOrSchemas);
349
542
  }
350
- export function createAdaptivePlaywrightRouter(routes) {
351
- return Router.create(routes);
543
+ /**
544
+ * An opt-in {@link AdaptivePlaywrightCrawlerOptions.resultComparator|`resultComparator`} that considers two
545
+ * request handler results equal only if *all* of their observable effects match - the pushed dataset items, the
546
+ * enqueued requests, and the key-value store changes. This is stricter than the default comparator, which only
547
+ * compares dataset items.
548
+ *
549
+ * **Beware:** enqueued URLs are compared exactly. The same page rendered in a browser and via plain HTTP often
550
+ * yields links that differ only in tracking query parameters, for example:
551
+ * - `https://sdk.apify.com/docs/guides/getting-started`
552
+ * - `https://sdk.apify.com/docs/guides/getting-started?__hsfp=1136113150&__hssc=7591405.1.173549427712`
553
+ *
554
+ * Such links are treated as *different*, which will make the crawler favor browser rendering for those pages.
555
+ *
556
+ * **Example usage:**
557
+ * ```ts
558
+ * const crawler = new AdaptivePlaywrightCrawler({
559
+ * resultComparator: fullResultComparator,
560
+ * async requestHandler({ pushData, enqueueLinks }) {
561
+ * // ...
562
+ * },
563
+ * });
564
+ * ```
565
+ */
566
+ export function fullResultComparator(resultA, resultB) {
567
+ return (isDeepStrictEqual(resultA.datasetItems, resultB.datasetItems) &&
568
+ isDeepStrictEqual(resultA.enqueuedUrls, resultB.enqueuedUrls) &&
569
+ isDeepStrictEqual(resultA.keyValueStoreChanges, resultB.keyValueStoreChanges));
352
570
  }
353
- //# sourceMappingURL=adaptive-playwright-crawler.js.map