@crawlee/playwright 4.0.0-beta.20 → 4.0.0-beta.201

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. package/README.md +14 -14
  2. package/index.d.ts +4 -3
  3. package/index.js +2 -2
  4. package/internals/adaptive-playwright-crawler.d.ts +136 -65
  5. package/internals/adaptive-playwright-crawler.js +416 -274
  6. package/internals/enqueue-links/click-elements.d.ts +36 -64
  7. package/internals/enqueue-links/click-elements.js +65 -67
  8. package/internals/playwright-browser-pool.d.ts +71 -0
  9. package/internals/playwright-browser-pool.js +61 -0
  10. package/internals/playwright-crawler.d.ts +176 -146
  11. package/internals/playwright-crawler.js +77 -73
  12. package/internals/playwright-launcher.d.ts +30 -20
  13. package/internals/playwright-launcher.js +22 -17
  14. package/internals/utils/playwright-utils.d.ts +54 -56
  15. package/internals/utils/playwright-utils.js +117 -138
  16. package/internals/utils/rendering-type-prediction.d.ts +37 -11
  17. package/internals/utils/rendering-type-prediction.js +81 -27
  18. package/package.json +15 -19
  19. package/index.d.ts.map +0 -1
  20. package/index.js.map +0 -1
  21. package/internals/adaptive-playwright-crawler.d.ts.map +0 -1
  22. package/internals/adaptive-playwright-crawler.js.map +0 -1
  23. package/internals/enqueue-links/click-elements.d.ts.map +0 -1
  24. package/internals/enqueue-links/click-elements.js.map +0 -1
  25. package/internals/playwright-crawler.d.ts.map +0 -1
  26. package/internals/playwright-crawler.js.map +0 -1
  27. package/internals/playwright-launcher.d.ts.map +0 -1
  28. package/internals/playwright-launcher.js.map +0 -1
  29. package/internals/utils/playwright-utils.d.ts.map +0 -1
  30. package/internals/utils/playwright-utils.js.map +0 -1
  31. package/internals/utils/rendering-type-prediction.d.ts.map +0 -1
  32. package/internals/utils/rendering-type-prediction.js.map +0 -1
@@ -20,19 +20,47 @@
20
20
  import { readFile } from 'node:fs/promises';
21
21
  import { createRequire } from 'node:module';
22
22
  import vm from 'node:vm';
23
- import { Configuration, KeyValueStore, SessionError, validators } from '@crawlee/browser';
24
- import { expandShadowRoots, sleep } from '@crawlee/utils';
25
- import * as cheerio from 'cheerio';
26
- import ow from 'ow';
23
+ import { Configuration, KeyValueStore, serviceLocator, SessionError, validators } from '@crawlee/browser';
24
+ import { sleep } from '@crawlee/utils';
25
+ import { expandShadowRoots, parseArgument, schemas } from '@crawlee/utils/internal';
26
+ import { z } from 'zod';
27
27
  import { LruCache } from '@apify/datastructures';
28
- import log_ from '@apify/log';
29
28
  import { enqueueLinksByClickingElements } from '../enqueue-links/click-elements.js';
30
- import { RenderingTypePredictor } from './rendering-type-prediction.js';
31
- const log = log_.child({ prefix: 'Playwright Utils' });
29
+ const getLog = () => serviceLocator.getChildLog('Playwright Utils');
32
30
  const require = createRequire(import.meta.url);
33
31
  const jqueryPath = require.resolve('jquery');
34
32
  const MAX_INJECT_FILE_CACHE_SIZE = 10;
35
33
  const DEFAULT_BLOCK_REQUEST_URL_PATTERNS = ['.css', '.jpg', '.jpeg', '.png', '.svg', '.gif', '.woff', '.pdf', '.zip'];
34
+ const filePathSchema = z.string();
35
+ const injectFileOptionsSchema = z.strictObject({
36
+ surviveNavigations: z.boolean().optional(),
37
+ });
38
+ const gotoExtendedRequestSchema = z.looseObject({
39
+ url: z.url(),
40
+ method: z.string().optional(),
41
+ headers: schemas.anyObject.optional(),
42
+ payload: z.union([z.string(), z.instanceof(Uint8Array)]).optional(),
43
+ });
44
+ const blockRequestsOptionsSchema = z.strictObject({
45
+ urlPatterns: schemas.arrayOf(z.string(), 'strings').default(DEFAULT_BLOCK_REQUEST_URL_PATTERNS),
46
+ extraUrlPatterns: schemas.arrayOf(z.string(), 'strings').default(() => []),
47
+ });
48
+ const infiniteScrollOptionsSchema = z.strictObject({
49
+ timeoutSecs: schemas.anyNumber.default(0),
50
+ maxScrollHeight: schemas.anyNumber.default(0),
51
+ waitForSecs: schemas.anyNumber.default(4),
52
+ scrollDownAndUp: z.boolean().default(false),
53
+ buttonSelector: z.string().optional(),
54
+ stopScrollCallback: schemas.anyFunction.optional(),
55
+ });
56
+ const saveSnapshotOptionsSchema = z.strictObject({
57
+ key: z.string().min(1).default('SNAPSHOT'),
58
+ screenshotQuality: schemas.anyNumber.default(50),
59
+ saveScreenshot: z.boolean().default(true),
60
+ saveHtml: z.boolean().default(true),
61
+ keyValueStoreName: z.string().optional(),
62
+ configuration: schemas.anyObject.optional(),
63
+ });
36
64
  /**
37
65
  * Cache contents of previously injected files to limit file system access.
38
66
  */
@@ -49,21 +77,19 @@ const injectedFilesCache = new LruCache({ maxLength: MAX_INJECT_FILE_CACHE_SIZE
49
77
  * @param [options]
50
78
  */
51
79
  export async function injectFile(page, filePath, options = {}) {
52
- ow(page, ow.object.validate(validators.browserPage));
53
- ow(filePath, ow.string);
54
- ow(options, ow.object.exactShape({
55
- surviveNavigations: ow.optional.boolean,
56
- }));
80
+ parseArgument(page, validators.browserPage);
81
+ parseArgument(filePath, filePathSchema);
82
+ const { surviveNavigations } = parseArgument(options, injectFileOptionsSchema);
57
83
  let contents = injectedFilesCache.get(filePath);
58
84
  if (!contents) {
59
85
  contents = await readFile(filePath, 'utf8');
60
86
  injectedFilesCache.add(filePath, contents);
61
87
  }
62
88
  const evalP = page.evaluate(contents);
63
- if (options.surviveNavigations) {
89
+ if (surviveNavigations) {
64
90
  page.on('framenavigated', async () => page
65
91
  .evaluate(contents)
66
- .catch((error) => log.warning('An error occurred during the script injection!', { error })));
92
+ .catch((error) => getLog().warning('An error occurred during the script injection!', { error })));
67
93
  }
68
94
  return evalP;
69
95
  }
@@ -94,7 +120,7 @@ export async function injectFile(page, filePath, options = {}) {
94
120
  * @param [options.surviveNavigations] Opt-out option to disable the JQuery reinjection after navigation.
95
121
  */
96
122
  export async function injectJQuery(page, options) {
97
- ow(page, ow.object.validate(validators.browserPage));
123
+ parseArgument(page, validators.browserPage);
98
124
  return injectFile(page, jqueryPath, { surviveNavigations: options?.surviveNavigations ?? true });
99
125
  }
100
126
  /**
@@ -110,19 +136,14 @@ export async function injectJQuery(page, options) {
110
136
  * @param [gotoOptions] Custom options for `page.goto()`.
111
137
  */
112
138
  export async function gotoExtended(page, request, gotoOptions = {}) {
113
- ow(page, ow.object.validate(validators.browserPage));
114
- ow(request, ow.object.partialShape({
115
- url: ow.string.url,
116
- method: ow.optional.string,
117
- headers: ow.optional.object,
118
- payload: ow.optional.any(ow.string, ow.uint8Array),
119
- }));
120
- ow(gotoOptions, ow.object);
139
+ parseArgument(page, validators.browserPage);
140
+ parseArgument(request, gotoExtendedRequestSchema);
141
+ parseArgument(gotoOptions, schemas.anyObject);
121
142
  const { url, method, headers, payload } = request;
122
143
  const isEmpty = (o) => !o || Object.keys(o).length === 0;
123
- if (method !== 'GET' || payload || !isEmpty(headers)) {
144
+ if (method !== 'GET' || payload) {
124
145
  // This is not deprecated, we use it to log only once.
125
- log.deprecated('Using other request methods than GET, rewriting headers and adding payloads has a high impact on performance ' +
146
+ getLog().deprecated('Using other request methods than GET, rewriting headers and adding payloads has a high impact on performance ' +
126
147
  'in recent versions of Playwright. Use only when necessary.');
127
148
  let wasCalled = false;
128
149
  const interceptRequestHandler = async (route) => {
@@ -143,12 +164,15 @@ export async function gotoExtended(page, request, gotoOptions = {}) {
143
164
  await route.continue(overrides);
144
165
  }
145
166
  catch (error) {
146
- log.debug('Error inside request interceptor', { error });
167
+ getLog().debug('Error inside request interceptor', { error });
147
168
  }
148
169
  return undefined;
149
170
  };
150
171
  await page.route('**/*', interceptRequestHandler);
151
172
  }
173
+ else if (!isEmpty(headers)) {
174
+ await page.setExtraHTTPHeaders(headers);
175
+ }
152
176
  return page.goto(url, gotoOptions);
153
177
  }
154
178
  /**
@@ -199,12 +223,8 @@ export async function gotoExtended(page, request, gotoOptions = {}) {
199
223
  * @param [options]
200
224
  */
201
225
  export async function blockRequests(page, options = {}) {
202
- ow(page, ow.object.validate(validators.browserPage));
203
- ow(options, ow.object.exactShape({
204
- urlPatterns: ow.optional.array.ofType(ow.string),
205
- extraUrlPatterns: ow.optional.array.ofType(ow.string),
206
- }));
207
- const { urlPatterns = DEFAULT_BLOCK_REQUEST_URL_PATTERNS, extraUrlPatterns = [] } = options;
226
+ parseArgument(page, validators.browserPage);
227
+ const { urlPatterns, extraUrlPatterns } = parseArgument(options, blockRequestsOptionsSchema);
208
228
  const patternsToBlock = [...urlPatterns, ...extraUrlPatterns];
209
229
  try {
210
230
  const client = await page.context().newCDPSession(page);
@@ -212,7 +232,7 @@ export async function blockRequests(page, options = {}) {
212
232
  await client.send('Network.setBlockedURLs', { urls: patternsToBlock });
213
233
  }
214
234
  catch {
215
- log.warning('blockRequests() helper is incompatible with non-Chromium browsers.');
235
+ getLog().warning('blockRequests() helper is incompatible with non-Chromium browsers.');
216
236
  }
217
237
  }
218
238
  /**
@@ -248,7 +268,7 @@ export function compileScript(scriptString, context = Object.create(null)) {
248
268
  func = vm.runInNewContext(funcString, context); // "Secure" the context by removing prototypes, unless custom context is provided.
249
269
  }
250
270
  catch (err) {
251
- log.exception(err, 'Cannot compile script!');
271
+ getLog().exception(err, 'Cannot compile script!');
252
272
  throw err;
253
273
  }
254
274
  if (typeof func !== 'function')
@@ -262,16 +282,8 @@ export function compileScript(scriptString, context = Object.create(null)) {
262
282
  * @param [options]
263
283
  */
264
284
  export async function infiniteScroll(page, options = {}) {
265
- ow(page, ow.object.validate(validators.browserPage));
266
- ow(options, ow.object.exactShape({
267
- timeoutSecs: ow.optional.number,
268
- maxScrollHeight: ow.optional.number,
269
- waitForSecs: ow.optional.number,
270
- scrollDownAndUp: ow.optional.boolean,
271
- buttonSelector: ow.optional.string,
272
- stopScrollCallback: ow.optional.function,
273
- }));
274
- const { timeoutSecs = 0, maxScrollHeight = 0, waitForSecs = 4, scrollDownAndUp = false, buttonSelector, stopScrollCallback, } = options;
285
+ parseArgument(page, validators.browserPage);
286
+ const { timeoutSecs, maxScrollHeight, waitForSecs, scrollDownAndUp, buttonSelector, stopScrollCallback } = parseArgument(options, infiniteScrollOptionsSchema);
275
287
  let finished;
276
288
  const startTime = Date.now();
277
289
  const CHECK_INTERVAL_MILLIS = 1000;
@@ -348,19 +360,11 @@ export async function infiniteScroll(page, options = {}) {
348
360
  * @param [options]
349
361
  */
350
362
  export async function saveSnapshot(page, options = {}) {
351
- ow(page, ow.object.validate(validators.browserPage));
352
- ow(options, ow.object.exactShape({
353
- key: ow.optional.string.nonEmpty,
354
- screenshotQuality: ow.optional.number,
355
- saveScreenshot: ow.optional.boolean,
356
- saveHtml: ow.optional.boolean,
357
- keyValueStoreName: ow.optional.string,
358
- config: ow.optional.object,
359
- }));
360
- const { key = 'SNAPSHOT', screenshotQuality = 50, saveScreenshot = true, saveHtml = true, keyValueStoreName, config, } = options;
363
+ parseArgument(page, validators.browserPage);
364
+ const { key, screenshotQuality, saveScreenshot, saveHtml, keyValueStoreName, configuration } = parseArgument(options, saveSnapshotOptionsSchema);
361
365
  try {
362
- const store = await KeyValueStore.open(keyValueStoreName, {
363
- config: config ?? Configuration.getGlobalConfig(),
366
+ const store = await KeyValueStore.open(keyValueStoreName ? { name: keyValueStoreName } : null, {
367
+ configuration: configuration ?? Configuration.getGlobalConfiguration(),
364
368
  });
365
369
  if (saveScreenshot) {
366
370
  const screenshotName = `${key}.jpg`;
@@ -395,13 +399,26 @@ export async function saveSnapshot(page, options = {}) {
395
399
  * @param ignoreShadowRoots
396
400
  */
397
401
  export async function parseWithCheerio(page, ignoreShadowRoots = false, ignoreIframes = false) {
398
- ow(page, ow.object.validate(validators.browserPage));
402
+ parseArgument(page, validators.browserPage);
403
+ const html = ignoreShadowRoots
404
+ ? null
405
+ : (await page.evaluate(`(${expandShadowRoots.toString()})(document)`));
406
+ const pageContent = html || (await page.content());
407
+ // Full cheerio (parse5) import for browser compliant parser
408
+ const { load } = await import('cheerio');
409
+ const $ = load(pageContent);
399
410
  if (page.frames().length > 1 && !ignoreIframes) {
400
411
  const frames = await page.$$('iframe');
401
- await Promise.all(frames.map(async (frame) => {
412
+ const cheerioIframes = $('iframe').toArray();
413
+ if (frames.length !== cheerioIframes.length) {
414
+ serviceLocator
415
+ .getLogger()
416
+ .warning(`parseWithCheerio: iframe count mismatch between live DOM (${frames.length}) and page snapshot (${cheerioIframes.length}). Some iframes may not be expanded.`);
417
+ }
418
+ await Promise.all(frames.map(async (frame, index) => {
402
419
  try {
403
420
  const iframe = await frame.contentFrame();
404
- if (iframe) {
421
+ if (iframe && cheerioIframes[index]) {
405
422
  const getIframeHTML = async () => {
406
423
  try {
407
424
  return iframe.locator('body').first().innerHTML();
@@ -411,53 +428,15 @@ export async function parseWithCheerio(page, ignoreShadowRoots = false, ignoreIf
411
428
  }
412
429
  };
413
430
  const contents = await getIframeHTML();
414
- await frame.evaluate((f, c) => {
415
- const replacementNode = document.createElement('div');
416
- replacementNode.innerHTML = c;
417
- replacementNode.className = 'crawlee-iframe-replacement';
418
- f.replaceWith(replacementNode);
419
- }, contents);
431
+ $(cheerioIframes[index]).replaceWith(`<div class="crawlee-iframe-replacement">${contents}</div>`);
420
432
  }
421
433
  }
422
434
  catch (error) {
423
- log.warning(`Failed to extract iframe content: ${error}`);
435
+ getLog().warning(`Failed to extract iframe content: ${error}`);
424
436
  }
425
437
  }));
426
438
  }
427
- const html = ignoreShadowRoots
428
- ? null
429
- : (await page.evaluate(`(${expandShadowRoots.toString()})(document)`));
430
- const pageContent = html || (await page.content());
431
- return cheerio.load(pageContent);
432
- }
433
- let idcacPlaywright = null;
434
- async function getIdcacPlaywright() {
435
- if (idcacPlaywright)
436
- return idcacPlaywright;
437
- try {
438
- idcacPlaywright = await import('idcac-playwright');
439
- }
440
- catch (error) {
441
- log.warning(`Failed to import 'idcac-playwright'.
442
-
443
- We recently made idcac-playwright an optional dependency due to licensing issues.
444
- To use this feature, please install it manually by running
445
-
446
- npm install idcac-playwright
447
-
448
- Original error message follows:
449
-
450
- ${error.message}
451
- `);
452
- }
453
- return idcacPlaywright;
454
- }
455
- export async function closeCookieModals(page) {
456
- ow(page, ow.object.validate(validators.browserPage));
457
- const idcac = await getIdcacPlaywright();
458
- if (idcac?.getInjectableScript()) {
459
- await page.evaluate(idcac.getInjectableScript());
460
- }
439
+ return $;
461
440
  }
462
441
  /**
463
442
  * This helper tries to solve the Cloudflare challenge automatically by clicking on the checkbox.
@@ -466,30 +445,24 @@ export async function closeCookieModals(page) {
466
445
  * result in a SessionError which will be automatically retried, so only successful requests will get
467
446
  * into the `requestHandler`.
468
447
  *
448
+ * On a successfully solved challenge the page is reloaded and the new {@link Response} is returned, so
449
+ * it can be propagated back to the crawling context via a hook return value (see
450
+ * {@link handleCloudflareChallengeHook}).
451
+ *
469
452
  * Works best with camoufox.
470
453
  *
471
454
  * **Example usage**
472
455
  * ```ts
473
456
  * postNavigationHooks: [
474
- * async ({ handleCloudflareChallenge }) => {
475
- * await handleCloudflareChallenge();
476
- * },
457
+ * async (context) => ({ response: await context.handleCloudflareChallenge() }),
477
458
  * ],
478
459
  * ```
479
460
  *
480
461
  * @param page Playwright [`Page`](https://playwright.dev/docs/api/class-page) object
481
462
  * @param url current URL for request identification, only used for logging
482
- * @param [session] current session object
483
463
  * @param [options]
484
464
  */
485
- async function handleCloudflareChallenge(page, url, session, options = {}) {
486
- // eslint-disable-next-line dot-notation
487
- const blockedStatusCodes = session?.['sessionPool']['blockedStatusCodes'];
488
- // Cloudflare pages are 403, which are blocked by default
489
- if (blockedStatusCodes?.includes(403)) {
490
- const idx = blockedStatusCodes.indexOf(403);
491
- blockedStatusCodes.splice(idx, 1);
492
- }
465
+ export async function handleCloudflareChallenge(page, url, options = {}) {
493
466
  options.isBlockedCallback ??= async () => {
494
467
  const isBlocked = await page.evaluate(() => {
495
468
  return document.querySelector('h1')?.textContent?.trim().includes('Sorry, you have been blocked');
@@ -498,7 +471,9 @@ async function handleCloudflareChallenge(page, url, session, options = {}) {
498
471
  };
499
472
  options.isChallengeCallback ??= async () => {
500
473
  return await page.evaluate(async () => {
501
- return !!document.querySelector('.footer > .footer-inner > .diagnostic-wrapper > .ray-id');
474
+ // Cloudflare keeps reshuffling the wrapper elements between `.footer-inner` and `.ray-id`,
475
+ // so only the stable outer classes are matched.
476
+ return !!document.querySelector('.footer .footer-inner .ray-id');
502
477
  });
503
478
  };
504
479
  const retryBlocked = async () => {
@@ -509,35 +484,49 @@ async function handleCloudflareChallenge(page, url, session, options = {}) {
509
484
  };
510
485
  // check if we ended up on the CF challenge page
511
486
  const isChallenge = async () => {
512
- return options.isChallengeCallback(page).catch(() => false);
487
+ return options.isChallengeCallback(page).catch(async () => {
488
+ // the challenge page navigates itself, which destroys the context of an in-flight check
489
+ await page.waitForLoadState('domcontentloaded').catch(() => { });
490
+ return options.isChallengeCallback(page).catch(() => false);
491
+ });
513
492
  };
514
493
  if (!(await isChallenge())) {
515
494
  await retryBlocked();
516
- return;
495
+ return undefined;
517
496
  }
518
497
  const logLevel = options.verbose ? 'info' : 'debug';
519
- log[logLevel](`Detected Cloudflare challenge at ${url}, trying to solve it. This can take up to ${10 + (options.sleepSecs ?? 10)} seconds.`);
498
+ getLog()[logLevel](`Detected Cloudflare challenge at ${url}, trying to solve it. This can take up to ${10 + (options.sleepSecs ?? 10)} seconds.`);
520
499
  const bb = await page
521
500
  .evaluate(() => {
522
- const div = document.querySelector('.main-content div');
501
+ // Prefer the actual challenge widget (the box holding the Turnstile checkbox input);
502
+ // fall back to the first content div for older challenge layouts.
503
+ const div = document.querySelector('.main-content div:has(input[id^="cf-chl-widget-"])') ??
504
+ document.querySelector('.main-content div');
523
505
  return div?.getBoundingClientRect();
524
506
  })
525
507
  .catch(() => undefined);
526
508
  if (!bb) {
527
- return;
509
+ return undefined;
528
510
  }
529
511
  const randomOffset = (range) => {
530
512
  return Math.round(100 * range * Math.random()) / 100;
531
513
  };
532
- const x = bb.x + 30;
533
- const y = bb.y + 25;
514
+ let x = bb.x + 30;
515
+ let y = bb.y + 25;
534
516
  // try to click the checkbox every second
535
517
  for (let i = 0; i < 10; i++) {
536
- await sleep(1000);
518
+ await sleep((options.preChallengeSleepSecs ?? 1) * 1000);
537
519
  // break early if we are no longer on the CF challenge page
538
520
  if (!(await isChallenge())) {
539
521
  break;
540
522
  }
523
+ if (options.clickPositionCallback) {
524
+ const pos = await options.clickPositionCallback(page);
525
+ if (pos) {
526
+ x = pos.x;
527
+ y = pos.y;
528
+ }
529
+ }
541
530
  if (options.clickCallback) {
542
531
  await options.clickCallback(page, { x, y });
543
532
  continue;
@@ -545,7 +534,10 @@ async function handleCloudflareChallenge(page, url, session, options = {}) {
545
534
  // we can click on the text too, so X can be a bit larger
546
535
  const xRandomized = x + randomOffset(10);
547
536
  const yRandomized = y + randomOffset(10);
548
- log[logLevel](`Trying to click on the Cloudflare checkbox at ${url}`, { x: xRandomized, y: yRandomized });
537
+ getLog()[logLevel](`Trying to click on the Cloudflare checkbox at ${url}`, {
538
+ x: xRandomized,
539
+ y: yRandomized,
540
+ });
549
541
  await page.mouse.click(xRandomized, yRandomized);
550
542
  // sometimes the checkbox is lower (could be caused by a lag when rendering the logo)
551
543
  await page.mouse.click(xRandomized, yRandomized + 35);
@@ -555,21 +547,8 @@ async function handleCloudflareChallenge(page, url, session, options = {}) {
555
547
  throw new SessionError(`Blocked by Cloudflare when processing ${url}`);
556
548
  }
557
549
  await retryBlocked();
550
+ // Reload to obtain a fresh Response without the challenge interstitial, which the caller can
551
+ // propagate back into the crawling context so downstream status-code checks see the new value.
552
+ return (await page.reload()) ?? undefined;
558
553
  }
559
554
  export { enqueueLinksByClickingElements };
560
- /** @internal */
561
- export const playwrightUtils = {
562
- injectFile,
563
- injectJQuery,
564
- gotoExtended,
565
- blockRequests,
566
- enqueueLinksByClickingElements,
567
- parseWithCheerio,
568
- infiniteScroll,
569
- saveSnapshot,
570
- compileScript,
571
- closeCookieModals,
572
- RenderingTypePredictor,
573
- handleCloudflareChallenge,
574
- };
575
- //# sourceMappingURL=playwright-utils.js.map
@@ -1,26 +1,54 @@
1
1
  import type { RecoverableStatePersistenceOptions, Request } from '@crawlee/core';
2
+ import type { Awaitable } from '@crawlee/types';
2
3
  export type RenderingType = 'clientOnly' | 'static';
3
- type URLComponents = string[];
4
- type FeatureVector = [staticResultsSimilarity: number, clientOnlyResultsSimilarity: number];
5
4
  export interface RenderingTypePredictorOptions {
6
5
  /** A number between 0 and 1 that determines the desired ratio of rendering type detections */
7
6
  detectionRatio: number;
8
7
  persistenceOptions?: Partial<RecoverableStatePersistenceOptions>;
9
8
  }
9
+ /**
10
+ * Minimal contract that any object passed to {@link AdaptivePlaywrightCrawler} as its
11
+ * `renderingTypePredictor` option must satisfy.
12
+ *
13
+ * @experimental
14
+ */
15
+ export interface IRenderingTypePredictor {
16
+ /**
17
+ * Predict the rendering type for a request, and how likely the crawler should be to verify it.
18
+ *
19
+ * Called once per request, before navigation - the crawler awaits it, so prefer loading whatever the
20
+ * prediction needs up front over per-request I/O.
21
+ */
22
+ predict(request: Request): Awaitable<{
23
+ renderingType: RenderingType;
24
+ detectionProbabilityRecommendation: number;
25
+ }>;
26
+ /**
27
+ * Report a detected rendering type, so that future predictions can take it into account.
28
+ *
29
+ * The crawler does not await this per detection - it collects the returned promises and drains them in
30
+ * `teardown()`, so an implementation that persists results asynchronously can keep batching its writes
31
+ * and still be sure they land before the crawl ends.
32
+ */
33
+ storeResult(requests: Request | Request[], renderingType: RenderingType): Awaitable<void>;
34
+ }
10
35
  /**
11
36
  * Stores rendering type information for previously crawled URLs and predicts the rendering type for URLs that have yet to be crawled and recommends when rendering type detection should be performed.
12
37
  *
13
38
  * @experimental
14
39
  */
15
- export declare class RenderingTypePredictor {
16
- private renderingTypeDetectionResults;
17
- private detectionRatio;
18
- private state;
40
+ export declare class RenderingTypePredictor implements IRenderingTypePredictor {
41
+ #private;
19
42
  constructor({ detectionRatio, persistenceOptions }: RenderingTypePredictorOptions);
20
43
  /**
21
44
  * Initialize the predictor by restoring persisted state.
22
45
  */
23
46
  initialize(): Promise<void>;
47
+ /**
48
+ * Stop persisting the model, writing it out one last time. `initialize()` reopens the persistence window.
49
+ */
50
+ teardown(): Promise<void>;
51
+ [Symbol.asyncDispose](): Promise<void>;
24
52
  /**
25
53
  * Predict the rendering type for a given URL and request label.
26
54
  */
@@ -31,10 +59,8 @@ export declare class RenderingTypePredictor {
31
59
  /**
32
60
  * Store the rendering type for a given URL and request label. This updates the underlying prediction model, which may be costly.
33
61
  */
34
- storeResult({ url, loadedUrl, label }: Request, renderingType: RenderingType): void;
62
+ storeResult(requests: Request | Request[], renderingType: RenderingType): void;
35
63
  private resultCount;
36
- protected calculateFeatureVector(url: URLComponents, label: string | undefined): FeatureVector;
37
- protected retrain(): void;
64
+ private calculateFeatureVector;
65
+ private retrain;
38
66
  }
39
- export {};
40
- //# sourceMappingURL=rendering-type-prediction.d.ts.map