@crawlee/cheerio 4.0.0-beta.16 → 4.0.0-beta.161

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -1,23 +1,23 @@
1
1
  <h1 align="center">
2
2
  <a href="https://crawlee.dev">
3
3
  <picture>
4
- <source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-dark.svg?sanitize=true">
5
- <img alt="Crawlee" src="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-light.svg?sanitize=true" width="500">
4
+ <source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-dark.svg?sanitize=true" />
5
+ <img alt="Crawlee" src="https://raw.githubusercontent.com/apify/crawlee/master/website/static/img/crawlee-light.svg?sanitize=true" width="500" />
6
6
  </picture>
7
7
  </a>
8
- <br>
8
+ <br />
9
9
  <small>A web scraping and browser automation library</small>
10
10
  </h1>
11
11
 
12
- <p align=center>
13
- <a href="https://trendshift.io/repositories/5179" target="_blank"><img src="https://trendshift.io/api/badge/repositories/5179" alt="apify%2Fcrawlee | Trendshift" style="width: 250px; height: 55px;" width="250" height="55"/></a>
12
+ <p align="center">
13
+ <a href="https://trendshift.io/repositories/5179" target="_blank"><img src="https://trendshift.io/api/badge/repositories/5179" alt="apify%2Fcrawlee | Trendshift" width="250" height="55"/></a>
14
14
  </p>
15
15
 
16
- <p align=center>
17
- <a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/v/@crawlee/core.svg" alt="NPM latest version" data-canonical-src="https://img.shields.io/npm/v/@crawlee/core/next.svg" style="max-width: 100%;"></a>
18
- <a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/dm/@crawlee/core.svg" alt="Downloads" data-canonical-src="https://img.shields.io/npm/dm/@crawlee/core.svg" style="max-width: 100%;"></a>
19
- <a href="https://discord.gg/jyEM2PRvMU" rel="nofollow"><img src="https://img.shields.io/discord/801163717915574323?label=discord" alt="Chat on discord" data-canonical-src="https://img.shields.io/discord/801163717915574323?label=discord" style="max-width: 100%;"></a>
20
- <a href="https://github.com/apify/crawlee/actions/workflows/test-ci.yml"><img src="https://github.com/apify/crawlee/actions/workflows/test-ci.yml/badge.svg?branch=master" alt="Build Status" style="max-width: 100%;"></a>
16
+ <p align="center">
17
+ <a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/v/@crawlee/core.svg" alt="NPM latest version" data-canonical-src="https://img.shields.io/npm/v/@crawlee/core/next.svg" /></a>
18
+ <a href="https://www.npmjs.com/package/@crawlee/core" rel="nofollow"><img src="https://img.shields.io/npm/dm/@crawlee/core.svg" alt="Downloads" data-canonical-src="https://img.shields.io/npm/dm/@crawlee/core.svg" /></a>
19
+ <a href="https://discord.gg/jyEM2PRvMU" rel="nofollow"><img src="https://img.shields.io/discord/801163717915574323?label=discord" alt="Chat on discord" data-canonical-src="https://img.shields.io/discord/801163717915574323?label=discord" /></a>
20
+ <a href="https://github.com/apify/crawlee/actions/workflows/test-ci.yml"><img src="https://github.com/apify/crawlee/actions/workflows/test-ci.yml/badge.svg?branch=master" alt="Build Status" /></a>
21
21
  </p>
22
22
 
23
23
  Crawlee covers your crawling and scraping end-to-end and **helps you build reliable scrapers. Fast.**
@@ -89,7 +89,7 @@ By default, Crawlee stores data to `./storage` in the current working directory.
89
89
  We provide automated beta builds for every merged code change in Crawlee. You can find them in the npm [list of releases](https://www.npmjs.com/package/crawlee?activeTab=versions). If you want to test new features or bug fixes before we release them, feel free to install a beta build like this:
90
90
 
91
91
  ```bash
92
- npm install crawlee@3.12.3-beta.13
92
+ npm install crawlee@next
93
93
  ```
94
94
 
95
95
  If you also use the [Apify SDK](https://github.com/apify/apify-sdk-js), you need to specify dependency overrides in your `package.json` file so that you don't end up with multiple versions of Crawlee installed:
@@ -98,9 +98,9 @@ If you also use the [Apify SDK](https://github.com/apify/apify-sdk-js), you need
98
98
  {
99
99
  "overrides": {
100
100
  "apify": {
101
- "@crawlee/core": "3.12.3-beta.13",
102
- "@crawlee/types": "3.12.3-beta.13",
103
- "@crawlee/utils": "3.12.3-beta.13"
101
+ "@crawlee/core": "$crawlee",
102
+ "@crawlee/types": "$crawlee",
103
+ "@crawlee/utils": "$crawlee"
104
104
  }
105
105
  }
106
106
  }
package/index.d.ts CHANGED
@@ -1,3 +1,2 @@
1
1
  export * from '@crawlee/http';
2
2
  export * from './internals/cheerio-crawler.js';
3
- //# sourceMappingURL=index.d.ts.map
package/index.js CHANGED
@@ -1,3 +1,2 @@
1
1
  export * from '@crawlee/http';
2
2
  export * from './internals/cheerio-crawler.js';
3
- //# sourceMappingURL=index.js.map
@@ -1,12 +1,14 @@
1
- import type { BasicCrawlingContext, Configuration, EnqueueLinksOptions, ErrorHandler, GetUserDataFromRequest, HttpCrawlerOptions, InternalHttpCrawlingContext, InternalHttpHook, RequestHandler, RequestProvider, RouterRoutes, SkippedRequestCallback } from '@crawlee/http';
1
+ import type { AddRequestsBatchedResult, ContextPipeline, CrawlingContext, EnqueueLinksOptions, ErrorHandler, ExtractLinksOptions, GetUserDataFromRequest, HttpCrawlerOptions, InternalHttpCrawlingContext, InternalHttpHook, RequestHandler, RouterHandler, RouterRoutes, RouteSchemas, RoutesFromSchemas } from '@crawlee/http';
2
2
  import { HttpCrawler } from '@crawlee/http';
3
- import type { BatchAddRequestsResult, Dictionary } from '@crawlee/types';
4
- import { type CheerioRoot, type RobotsTxtFile } from '@crawlee/utils';
3
+ import type { Dictionary } from '@crawlee/types';
4
+ import type { CheerioAPI } from 'cheerio';
5
5
  import * as cheerio from 'cheerio';
6
6
  export type CheerioErrorHandler<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
7
- JSONData extends Dictionary = any> = ErrorHandler<CheerioCrawlingContext<UserData, JSONData>>;
7
+ JSONData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
8
+ ContextExtension = Dictionary<never>> = ErrorHandler<CrawlingContext, CheerioCrawlingContext<UserData, JSONData> & ContextExtension>;
8
9
  export interface CheerioCrawlerOptions<ContextExtension = Dictionary<never>, ExtendedContext extends CheerioCrawlingContext = CheerioCrawlingContext & ContextExtension, UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
9
- JSONData extends Dictionary = any> extends HttpCrawlerOptions<CheerioCrawlingContext<UserData, JSONData>, ContextExtension, ExtendedContext> {
10
+ JSONData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
11
+ Routes extends Record<keyof Routes, Dictionary> = Record<string, UserData>, StatisticStateExtension extends object = {}> extends HttpCrawlerOptions<CheerioCrawlingContext<UserData, JSONData>, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
10
12
  }
11
13
  export type CheerioHook<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
12
14
  JSONData extends Dictionary = any> = InternalHttpHook<CheerioCrawlingContext<UserData, JSONData>>;
@@ -48,11 +50,15 @@ JSONData extends Dictionary = any> extends InternalHttpCrawlingContext<UserData,
48
50
  * });
49
51
  * ```
50
52
  */
51
- parseWithCheerio(selector?: string, timeoutMs?: number): Promise<CheerioRoot>;
53
+ parseWithCheerio(selector?: string, timeoutMs?: number): Promise<CheerioAPI>;
54
+ /**
55
+ * Extracts URLs from the parsed HTML, without adding them to the request queue.
56
+ */
57
+ extractLinks(options?: ExtractLinksOptions): Promise<string[]>;
52
58
  /**
53
59
  * Helper function for extracting URLs from the parsed HTML and adding them to the request queue.
54
60
  */
55
- enqueueLinks(options?: EnqueueLinksOptions): Promise<BatchAddRequestsResult>;
61
+ enqueueLinks(options?: EnqueueLinksOptions): Promise<AddRequestsBatchedResult>;
56
62
  }
57
63
  export type CheerioRequestHandler<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
58
64
  JSONData extends Dictionary = any> = RequestHandler<CheerioCrawlingContext<UserData, JSONData>>;
@@ -72,38 +78,40 @@ JSONData extends Dictionary = any> = RequestHandler<CheerioCrawlingContext<UserD
72
78
  * and then invokes the user-provided {@link CheerioCrawlerOptions.requestHandler} to extract page data
73
79
  * using a [jQuery](https://jquery.com/)-like interface to the parsed HTML DOM.
74
80
  *
75
- * The source URLs are represented using {@link Request} objects that are fed from
76
- * {@link RequestList} or {@link RequestQueue} instances provided by the {@link CheerioCrawlerOptions.requestList}
77
- * or {@link CheerioCrawlerOptions.requestQueue} constructor options, respectively.
81
+ * The source URLs are represented using {@link Request} objects that are fed from the
82
+ * {@link IRequestManager|request manager} provided via the {@link CheerioCrawlerOptions.requestManager|`requestManager`}
83
+ * constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
84
+ * as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
85
+ * {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
86
+ * result as `requestManager`.
78
87
  *
79
- * If both {@link CheerioCrawlerOptions.requestList} and {@link CheerioCrawlerOptions.requestQueue} are used,
80
- * the instance first processes URLs from the {@link RequestList} and automatically enqueues all of them
81
- * to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
88
+ * > The {@link CheerioCrawlerOptions.requestList|`requestList`} and {@link CheerioCrawlerOptions.requestQueue|`requestQueue`}
89
+ * > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
82
90
  *
83
91
  * The crawler finishes when there are no more {@link Request} objects to crawl.
84
92
  *
85
- * We can use the `preNavigationHooks` to adjust `gotOptions`:
93
+ * We can use the `preNavigationHooks` to adjust the crawling context before the request is made:
86
94
  *
87
95
  * ```
88
96
  * preNavigationHooks: [
89
- * (crawlingContext, gotOptions) => {
97
+ * (crawlingContext) => {
90
98
  * // ...
91
99
  * },
92
100
  * ]
93
101
  * ```
94
102
  *
95
- * By default, `CheerioCrawler` only processes web pages with the `text/html`
96
- * and `application/xhtml+xml` MIME content types (as reported by the `Content-Type` HTTP header),
103
+ * By default, `CheerioCrawler` only processes web pages with the `text/html`, `application/xhtml+xml`, `text/xml`, `application/xml`,
104
+ * and `application/json` MIME content types (as reported by the `Content-Type` HTTP header),
97
105
  * and skips pages with other content types. If you want the crawler to process other content types,
98
106
  * use the {@link CheerioCrawlerOptions.additionalMimeTypes} constructor option.
99
107
  * Beware that the parsing behavior differs for HTML, XML, JSON and other types of content.
100
108
  * For more details, see {@link CheerioCrawlerOptions.requestHandler}.
101
109
  *
102
- * New requests are only dispatched when there is enough free CPU and memory available,
103
- * using the functionality provided by the {@link AutoscaledPool} class.
104
- * All {@link AutoscaledPool} configuration options can be passed to the `autoscaledPoolOptions`
105
- * parameter of the `CheerioCrawler` constructor. For user convenience, the `minConcurrency` and `maxConcurrency`
106
- * {@link AutoscaledPool} options are available directly in the `CheerioCrawler` constructor.
110
+ * New requests are only dispatched when there is enough free CPU and memory available, as judged by the crawler's
111
+ * {@link ConcurrencySystem}.
112
+ * Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
113
+ * `CheerioCrawler` constructor, or, for finer control, by injecting a pre-configured
114
+ * {@link ConcurrencySystem|`concurrencySystem`}.
107
115
  *
108
116
  * **Example usage:**
109
117
  *
@@ -133,32 +141,15 @@ JSONData extends Dictionary = any> = RequestHandler<CheerioCrawlingContext<UserD
133
141
  * ```
134
142
  * @category Crawlers
135
143
  */
136
- export declare class CheerioCrawler<ContextExtension = Dictionary<never>, ExtendedContext extends CheerioCrawlingContext = CheerioCrawlingContext & ContextExtension> extends HttpCrawler<CheerioCrawlingContext, ContextExtension, ExtendedContext> {
144
+ export declare class CheerioCrawler<ContextExtension = Dictionary<never>, ExtendedContext extends CheerioCrawlingContext = CheerioCrawlingContext & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<CheerioCrawlingContext['request']>>, StatisticStateExtension extends object = {}> extends HttpCrawler<CheerioCrawlingContext, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
137
145
  /**
138
146
  * All `CheerioCrawler` parameters are passed via an options object.
139
147
  */
140
- constructor(options?: CheerioCrawlerOptions<ContextExtension, ExtendedContext>, config?: Configuration);
148
+ constructor(options?: CheerioCrawlerOptions<ContextExtension, ExtendedContext, any, any, Routes, StatisticStateExtension>);
149
+ protected buildContextPipeline(): ContextPipeline<CrawlingContext, CheerioCrawlingContext>;
141
150
  private parseContent;
142
151
  private addHelpers;
143
152
  }
144
- interface EnqueueLinksInternalOptions {
145
- options?: EnqueueLinksOptions;
146
- $: cheerio.CheerioAPI | null;
147
- requestQueue: RequestProvider;
148
- robotsTxtFile?: RobotsTxtFile;
149
- onSkippedRequest?: SkippedRequestCallback;
150
- originalRequestUrl: string;
151
- finalRequestUrl?: string;
152
- }
153
- interface BoundEnqueueLinksInternalOptions {
154
- enqueueLinks: BasicCrawlingContext['enqueueLinks'];
155
- options?: EnqueueLinksOptions;
156
- $: cheerio.CheerioAPI | null;
157
- originalRequestUrl: string;
158
- finalRequestUrl?: string;
159
- }
160
- /** @internal */
161
- export declare function cheerioCrawlerEnqueueLinks(options: EnqueueLinksInternalOptions | BoundEnqueueLinksInternalOptions): Promise<unknown>;
162
153
  /**
163
154
  * Creates new {@link Router} instance that works based on request labels.
164
155
  * This instance can then serve as a `requestHandler` of your {@link CheerioCrawler}.
@@ -183,7 +174,6 @@ export declare function cheerioCrawlerEnqueueLinks(options: EnqueueLinksInternal
183
174
  * await crawler.run();
184
175
  * ```
185
176
  */
186
- // @ts-ignore optional peer dependency or compatibility with es2022
187
- export declare function createCheerioRouter<Context extends CheerioCrawlingContext = CheerioCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, UserData>): import("@crawlee/http").RouterHandler<Context>;
188
- export {};
189
- //# sourceMappingURL=cheerio-crawler.d.ts.map
177
+ export declare function createCheerioRouter<Context extends CheerioCrawlingContext = CheerioCrawlingContext, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>>(routes?: RouterRoutes<Context, Routes>): RouterHandler<Context, Routes>;
178
+ export declare function createCheerioRouter<Context extends CheerioCrawlingContext = CheerioCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, Record<string, UserData>>): RouterHandler<Context, Record<string, UserData>>;
179
+ export declare function createCheerioRouter<Context extends CheerioCrawlingContext = CheerioCrawlingContext, const Schemas extends RouteSchemas = RouteSchemas>(schemas: Schemas): RouterHandler<Context, RoutesFromSchemas<Schemas>>;
@@ -1,5 +1,5 @@
1
- import { enqueueLinks, HttpCrawler, resolveBaseUrlForEnqueueLinksFiltering, Router } from '@crawlee/http';
2
- import { extractUrlsFromCheerio } from '@crawlee/utils';
1
+ import { EnqueueStrategy, HttpCrawler, NavigationSkippedError, resolveBaseUrlForEnqueueLinksFiltering, Router, } from '@crawlee/http';
2
+ import { extractUrlsFromCheerio } from '@crawlee/utils/internal';
3
3
  import * as cheerio from 'cheerio';
4
4
  import { parseDocument } from 'htmlparser2';
5
5
  /**
@@ -18,38 +18,40 @@ import { parseDocument } from 'htmlparser2';
18
18
  * and then invokes the user-provided {@link CheerioCrawlerOptions.requestHandler} to extract page data
19
19
  * using a [jQuery](https://jquery.com/)-like interface to the parsed HTML DOM.
20
20
  *
21
- * The source URLs are represented using {@link Request} objects that are fed from
22
- * {@link RequestList} or {@link RequestQueue} instances provided by the {@link CheerioCrawlerOptions.requestList}
23
- * or {@link CheerioCrawlerOptions.requestQueue} constructor options, respectively.
21
+ * The source URLs are represented using {@link Request} objects that are fed from the
22
+ * {@link IRequestManager|request manager} provided via the {@link CheerioCrawlerOptions.requestManager|`requestManager`}
23
+ * constructor option (a {@link RequestQueue} is itself a request manager). To read from a read-only source such
24
+ * as a {@link RequestList} while still being able to enqueue new requests, combine it with a queue into a
25
+ * {@link RequestManagerTandem} via {@link IRequestLoader.toTandem|`requestLoader.toTandem()`} and pass the
26
+ * result as `requestManager`.
24
27
  *
25
- * If both {@link CheerioCrawlerOptions.requestList} and {@link CheerioCrawlerOptions.requestQueue} are used,
26
- * the instance first processes URLs from the {@link RequestList} and automatically enqueues all of them
27
- * to {@link RequestQueue} before it starts their processing. This ensures that a single URL is not crawled multiple times.
28
+ * > The {@link CheerioCrawlerOptions.requestList|`requestList`} and {@link CheerioCrawlerOptions.requestQueue|`requestQueue`}
29
+ * > options are deprecated; they are still accepted and folded into a single `requestManager` for back-compat.
28
30
  *
29
31
  * The crawler finishes when there are no more {@link Request} objects to crawl.
30
32
  *
31
- * We can use the `preNavigationHooks` to adjust `gotOptions`:
33
+ * We can use the `preNavigationHooks` to adjust the crawling context before the request is made:
32
34
  *
33
35
  * ```
34
36
  * preNavigationHooks: [
35
- * (crawlingContext, gotOptions) => {
37
+ * (crawlingContext) => {
36
38
  * // ...
37
39
  * },
38
40
  * ]
39
41
  * ```
40
42
  *
41
- * By default, `CheerioCrawler` only processes web pages with the `text/html`
42
- * and `application/xhtml+xml` MIME content types (as reported by the `Content-Type` HTTP header),
43
+ * By default, `CheerioCrawler` only processes web pages with the `text/html`, `application/xhtml+xml`, `text/xml`, `application/xml`,
44
+ * and `application/json` MIME content types (as reported by the `Content-Type` HTTP header),
43
45
  * and skips pages with other content types. If you want the crawler to process other content types,
44
46
  * use the {@link CheerioCrawlerOptions.additionalMimeTypes} constructor option.
45
47
  * Beware that the parsing behavior differs for HTML, XML, JSON and other types of content.
46
48
  * For more details, see {@link CheerioCrawlerOptions.requestHandler}.
47
49
  *
48
- * New requests are only dispatched when there is enough free CPU and memory available,
49
- * using the functionality provided by the {@link AutoscaledPool} class.
50
- * All {@link AutoscaledPool} configuration options can be passed to the `autoscaledPoolOptions`
51
- * parameter of the `CheerioCrawler` constructor. For user convenience, the `minConcurrency` and `maxConcurrency`
52
- * {@link AutoscaledPool} options are available directly in the `CheerioCrawler` constructor.
50
+ * New requests are only dispatched when there is enough free CPU and memory available, as judged by the crawler's
51
+ * {@link ConcurrencySystem}.
52
+ * Concurrency is tuned via the `minConcurrency`, `maxConcurrency` and `maxRequestsPerMinute` options of the
53
+ * `CheerioCrawler` constructor, or, for finer control, by injecting a pre-configured
54
+ * {@link ConcurrencySystem|`concurrencySystem`}.
53
55
  *
54
56
  * **Example usage:**
55
57
  *
@@ -83,44 +85,73 @@ export class CheerioCrawler extends HttpCrawler {
83
85
  /**
84
86
  * All `CheerioCrawler` parameters are passed via an options object.
85
87
  */
86
- constructor(options, config) {
88
+ constructor(options) {
89
+ const { contextPipelineBuilder, ...rest } = options ?? {};
87
90
  super({
88
- ...options,
89
- contextPipelineBuilder: () => this.buildContextPipeline()
90
- .compose({
91
- action: async (context) => await this.parseContent(context),
92
- })
93
- .compose({ action: async (context) => await this.addHelpers(context) }),
94
- }, config);
91
+ ...rest,
92
+ contextPipelineBuilder: contextPipelineBuilder ?? (() => this.buildContextPipeline()),
93
+ });
94
+ }
95
+ buildContextPipeline() {
96
+ return super
97
+ .buildContextPipeline()
98
+ .compose({
99
+ action: async (context) => await this.parseContent(context),
100
+ })
101
+ .compose({ action: async (context) => await this.addHelpers(context) });
95
102
  }
96
103
  async parseContent(crawlingContext) {
97
- const isXml = crawlingContext.contentType.type.includes('xml');
98
- const body = Buffer.isBuffer(crawlingContext.body)
99
- ? crawlingContext.body.toString(crawlingContext.contentType.encoding)
100
- : crawlingContext.body;
101
- const dom = parseDocument(body, { decodeEntities: true, xmlMode: isXml });
102
- const $ = cheerio.load(dom, {
103
- xml: { decodeEntities: true, xmlMode: isXml },
104
- });
105
- return {
106
- $,
107
- body,
108
- };
104
+ try {
105
+ const isXml = crawlingContext.contentType.type.includes('xml');
106
+ const body = Buffer.isBuffer(crawlingContext.body)
107
+ ? crawlingContext.body.toString(crawlingContext.contentType.encoding)
108
+ : crawlingContext.body;
109
+ const dom = parseDocument(body, { decodeEntities: true, xmlMode: isXml });
110
+ const $ = cheerio.load(dom, {
111
+ xml: { decodeEntities: true, xmlMode: isXml },
112
+ });
113
+ return {
114
+ $,
115
+ body,
116
+ };
117
+ }
118
+ catch (err) {
119
+ if (err instanceof NavigationSkippedError) {
120
+ return {
121
+ get body() {
122
+ throw new NavigationSkippedError('The `body` property is not available - `skipNavigation` was used', { cause: err });
123
+ },
124
+ get $() {
125
+ throw new NavigationSkippedError('The `$` property is not available - `skipNavigation` was used', { cause: err });
126
+ },
127
+ };
128
+ }
129
+ throw err;
130
+ }
109
131
  }
110
132
  async addHelpers(crawlingContext) {
111
- const originalEnqueueLinks = crawlingContext.enqueueLinks;
133
+ const addRequests = crawlingContext.addRequests;
134
+ const extractLinks = async (options) => {
135
+ if (!crawlingContext.$) {
136
+ throw new Error('Cannot extract links because the DOM is not available.');
137
+ }
138
+ return extractUrlsFromCheerio(crawlingContext.$, options?.selector ?? 'a', options?.baseUrl ?? crawlingContext.request.loadedUrl ?? crawlingContext.request.url);
139
+ };
112
140
  return {
113
- enqueueLinks: async (enqueueOptions) => {
114
- return (await cheerioCrawlerEnqueueLinks({
115
- options: { ...enqueueOptions, limit: this.calculateEnqueuedRequestLimit(enqueueOptions?.limit) },
116
- $: crawlingContext.$,
117
- requestQueue: await this.getRequestQueue(),
118
- robotsTxtFile: await this.getRobotsTxtFileForUrl(crawlingContext.request.url),
119
- onSkippedRequest: this.handleSkippedRequest,
120
- originalRequestUrl: crawlingContext.request.url,
141
+ extractLinks,
142
+ enqueueLinks: async (options = {}) => {
143
+ const baseUrl = resolveBaseUrlForEnqueueLinksFiltering({
144
+ enqueueStrategy: options.strategy,
121
145
  finalRequestUrl: crawlingContext.request.loadedUrl,
122
- enqueueLinks: originalEnqueueLinks,
123
- })); // TODO make this type safe
146
+ originalRequestUrl: crawlingContext.request.url,
147
+ userProvidedBaseUrl: options.baseUrl,
148
+ });
149
+ const urls = await extractLinks(options);
150
+ return addRequests(urls, {
151
+ ...options,
152
+ baseUrl,
153
+ strategy: options.strategy ?? EnqueueStrategy.SameHostname,
154
+ });
124
155
  },
125
156
  waitForSelector: async (selector, _timeoutMs) => {
126
157
  if (crawlingContext.$(selector).get().length === 0) {
@@ -136,64 +167,6 @@ export class CheerioCrawler extends HttpCrawler {
136
167
  };
137
168
  }
138
169
  }
139
- /** @internal */
140
- function containsEnqueueLinks(options) {
141
- return !!options.enqueueLinks;
142
- }
143
- /** @internal */
144
- export async function cheerioCrawlerEnqueueLinks(options) {
145
- const { options: enqueueLinksOptions, $, originalRequestUrl, finalRequestUrl } = options;
146
- if (!$) {
147
- throw new Error('Cannot enqueue links because the DOM is not available.');
148
- }
149
- const baseUrl = resolveBaseUrlForEnqueueLinksFiltering({
150
- enqueueStrategy: enqueueLinksOptions?.strategy,
151
- finalRequestUrl,
152
- originalRequestUrl,
153
- userProvidedBaseUrl: enqueueLinksOptions?.baseUrl,
154
- });
155
- const urls = extractUrlsFromCheerio($, enqueueLinksOptions?.selector ?? 'a', enqueueLinksOptions?.baseUrl ?? finalRequestUrl ?? originalRequestUrl);
156
- if (containsEnqueueLinks(options)) {
157
- return options.enqueueLinks({
158
- urls,
159
- baseUrl,
160
- ...enqueueLinksOptions,
161
- });
162
- }
163
- return enqueueLinks({
164
- requestQueue: options.requestQueue,
165
- robotsTxtFile: options.robotsTxtFile,
166
- onSkippedRequest: options.onSkippedRequest,
167
- urls,
168
- baseUrl,
169
- ...enqueueLinksOptions,
170
- });
171
- }
172
- /**
173
- * Creates new {@link Router} instance that works based on request labels.
174
- * This instance can then serve as a `requestHandler` of your {@link CheerioCrawler}.
175
- * Defaults to the {@link CheerioCrawlingContext}.
176
- *
177
- * > Serves as a shortcut for using `Router.create<CheerioCrawlingContext>()`.
178
- *
179
- * ```ts
180
- * import { CheerioCrawler, createCheerioRouter } from 'crawlee';
181
- *
182
- * const router = createCheerioRouter();
183
- * router.addHandler('label-a', async (ctx) => {
184
- * ctx.log.info('...');
185
- * });
186
- * router.addDefaultHandler(async (ctx) => {
187
- * ctx.log.info('...');
188
- * });
189
- *
190
- * const crawler = new CheerioCrawler({
191
- * requestHandler: router,
192
- * });
193
- * await crawler.run();
194
- * ```
195
- */
196
- export function createCheerioRouter(routes) {
197
- return Router.create(routes);
170
+ export function createCheerioRouter(routesOrSchemas) {
171
+ return Router.create(routesOrSchemas);
198
172
  }
199
- //# sourceMappingURL=cheerio-crawler.js.map
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@crawlee/cheerio",
3
- "version": "4.0.0-beta.16",
3
+ "version": "4.0.0-beta.161",
4
4
  "description": "The scalable web crawling and scraping library for JavaScript/Node.js. Enables development of data extraction and web automation jobs (not only) with headless Chrome and Puppeteer.",
5
5
  "engines": {
6
6
  "node": ">=22.0.0"
@@ -38,7 +38,7 @@
38
38
  },
39
39
  "homepage": "https://crawlee.dev",
40
40
  "scripts": {
41
- "build": "yarn clean && yarn compile && yarn copy",
41
+ "build": "pnpm clean && pnpm compile && pnpm copy",
42
42
  "clean": "rimraf ./dist",
43
43
  "compile": "tsc -p tsconfig.build.json",
44
44
  "copy": "tsx ../../scripts/copy.ts"
@@ -47,9 +47,9 @@
47
47
  "access": "public"
48
48
  },
49
49
  "dependencies": {
50
- "@crawlee/http": "4.0.0-beta.16",
51
- "@crawlee/types": "4.0.0-beta.16",
52
- "@crawlee/utils": "4.0.0-beta.16",
50
+ "@crawlee/http": "4.0.0-beta.161",
51
+ "@crawlee/types": "4.0.0-beta.161",
52
+ "@crawlee/utils": "4.0.0-beta.161",
53
53
  "cheerio": "^1.0.0",
54
54
  "htmlparser2": "^10.0.0",
55
55
  "tslib": "^2.8.1"
@@ -61,5 +61,5 @@
61
61
  }
62
62
  }
63
63
  },
64
- "gitHead": "65b235c9bdcf0521e0fbae05c77f4adaa89c45e0"
64
+ "gitHead": "dc353e294d5f4d2ee468325ef11dd7af44443be2"
65
65
  }
package/index.d.ts.map DELETED
@@ -1 +0,0 @@
1
- {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,cAAc,eAAe,CAAC;AAC9B,cAAc,gCAAgC,CAAC"}
package/index.js.map DELETED
@@ -1 +0,0 @@
1
- {"version":3,"file":"index.js","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,cAAc,eAAe,CAAC;AAC9B,cAAc,gCAAgC,CAAC"}
@@ -1 +0,0 @@
1
- {"version":3,"file":"cheerio-crawler.d.ts","sourceRoot":"","sources":["../../src/internals/cheerio-crawler.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EACR,oBAAoB,EACpB,aAAa,EACb,mBAAmB,EACnB,YAAY,EACZ,sBAAsB,EACtB,kBAAkB,EAClB,2BAA2B,EAC3B,gBAAgB,EAChB,cAAc,EACd,eAAe,EACf,YAAY,EACZ,sBAAsB,EACzB,MAAM,eAAe,CAAC;AACvB,OAAO,EAAgB,WAAW,EAAkD,MAAM,eAAe,CAAC;AAC1G,OAAO,KAAK,EAAE,sBAAsB,EAAE,UAAU,EAAE,MAAM,gBAAgB,CAAC;AACzE,OAAO,EAAE,KAAK,WAAW,EAA0B,KAAK,aAAa,EAAE,MAAM,gBAAgB,CAAC;AAE9F,OAAO,KAAK,OAAO,MAAM,SAAS,CAAC;AAGnC,MAAM,MAAM,mBAAmB,CAC3B,QAAQ,SAAS,UAAU,GAAG,GAAG,EAAE,2EAA2E;AAC9G,QAAQ,SAAS,UAAU,GAAG,GAAG,IACjC,YAAY,CAAC,sBAAsB,CAAC,QAAQ,EAAE,QAAQ,CAAC,CAAC,CAAC;AAE7D,MAAM,WAAW,qBAAqB,CAClC,gBAAgB,GAAG,UAAU,CAAC,KAAK,CAAC,EACpC,eAAe,SAAS,sBAAsB,GAAG,sBAAsB,GAAG,gBAAgB,EAC1F,QAAQ,SAAS,UAAU,GAAG,GAAG,EAAE,2EAA2E;AAC9G,QAAQ,SAAS,UAAU,GAAG,GAAG,CACnC,SAAQ,kBAAkB,CAAC,sBAAsB,CAAC,QAAQ,EAAE,QAAQ,CAAC,EAAE,gBAAgB,EAAE,eAAe,CAAC;CAAG;AAE9G,MAAM,MAAM,WAAW,CACnB,QAAQ,SAAS,UAAU,GAAG,GAAG,EAAE,2EAA2E;AAC9G,QAAQ,SAAS,UAAU,GAAG,GAAG,IACjC,gBAAgB,CAAC,sBAAsB,CAAC,QAAQ,EAAE,QAAQ,CAAC,CAAC,CAAC;AAEjE,MAAM,WAAW,sBAAsB,CACnC,QAAQ,SAAS,UAAU,GAAG,GAAG,EAAE,2EAA2E;AAC9G,QAAQ,SAAS,UAAU,GAAG,GAAG,CACnC,SAAQ,2BAA2B,CAAC,QAAQ,EAAE,QAAQ,CAAC;IACrD;;OAEG;IACH,IAAI,EAAE,MAAM,CAAC;IAEb;;;OAGG;IACH,CAAC,EAAE,OAAO,CAAC,UAAU,CAAC;IAEtB;;;;;;;;;;;OAWG;IACH,eAAe,CAAC,QAAQ,EAAE,MAAM,EAAE,SAAS,CAAC,EAAE,MAAM,GAAG,OAAO,CAAC,IAAI,CAAC,CAAC;IAErE;;;;;;;;;;;;;OAaG;IACH,gBAAgB,CAAC,QAAQ,CAAC,EAAE,MAAM,EAAE,SAAS,CAAC,EAAE,MAAM,GAAG,OAAO,CAAC,WAAW,CAAC,CAAC;IAE9E;;OAEG;IACH,YAAY,CAAC,OAAO,CAAC,EAAE,mBAAmB,GAAG,OAAO,CAAC,sBAAsB,CAAC,CAAC;CAChF;AAED,MAAM,MAAM,qBAAqB,CAC7B,QAAQ,SAAS,UAAU,GAAG,GAAG,EAAE,2EAA2E;AAC9G,QAAQ,SAAS,UAAU,GAAG,GAAG,IACjC,cAAc,CAAC,sBAAsB,CAAC,QAAQ,EAAE,QAAQ,CAAC,CAAC,CAAC;AAE/D;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA4EG;AACH,qBAAa,cAAc,CACvB,gBAAgB,GAAG,UAAU,CAAC,KAAK,CAAC,EACpC,eAAe,SAAS,sBAAsB,GAAG,sBAAsB,GAAG,gBAAgB,CAC5F,SAAQ,WAAW,CAAC,sBAAsB,EAAE,gBAAgB,EAAE,eAAe,CAAC;IAC5E;;OAEG;gBACS,OAAO,CAAC,EAAE,qBAAqB,CAAC,gBAAgB,EAAE,eAAe,CAAC,EAAE,MAAM,CAAC,EAAE,aAAa;YAexF,YAAY;YAgBZ,UAAU;CA8B3B;AAED,UAAU,2BAA2B;IACjC,OAAO,CAAC,EAAE,mBAAmB,CAAC;IAC9B,CAAC,EAAE,OAAO,CAAC,UAAU,GAAG,IAAI,CAAC;IAC7B,YAAY,EAAE,eAAe,CAAC;IAC9B,aAAa,CAAC,EAAE,aAAa,CAAC;IAC9B,gBAAgB,CAAC,EAAE,sBAAsB,CAAC;IAC1C,kBAAkB,EAAE,MAAM,CAAC;IAC3B,eAAe,CAAC,EAAE,MAAM,CAAC;CAC5B;AAED,UAAU,gCAAgC;IACtC,YAAY,EAAE,oBAAoB,CAAC,cAAc,CAAC,CAAC;IACnD,OAAO,CAAC,EAAE,mBAAmB,CAAC;IAC9B,CAAC,EAAE,OAAO,CAAC,UAAU,GAAG,IAAI,CAAC;IAC7B,kBAAkB,EAAE,MAAM,CAAC;IAC3B,eAAe,CAAC,EAAE,MAAM,CAAC;CAC5B;AASD,gBAAgB;AAChB,wBAAsB,0BAA0B,CAC5C,OAAO,EAAE,2BAA2B,GAAG,gCAAgC,oBAmC1E;AAED;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,wBAAgB,mBAAmB,CAC/B,OAAO,SAAS,sBAAsB,GAAG,sBAAsB,EAC/D,QAAQ,SAAS,UAAU,GAAG,sBAAsB,CAAC,OAAO,CAAC,SAAS,CAAC,CAAC,EAC1E,MAAM,CAAC,EAAE,YAAY,CAAC,OAAO,EAAE,QAAQ,CAAC,kDAEzC"}
@@ -1 +0,0 @@
1
- {"version":3,"file":"cheerio-crawler.js","sourceRoot":"","sources":["../../src/internals/cheerio-crawler.ts"],"names":[],"mappings":"AAcA,OAAO,EAAE,YAAY,EAAE,WAAW,EAAE,sCAAsC,EAAE,MAAM,EAAE,MAAM,eAAe,CAAC;AAE1G,OAAO,EAAoB,sBAAsB,EAAsB,MAAM,gBAAgB,CAAC;AAE9F,OAAO,KAAK,OAAO,MAAM,SAAS,CAAC;AACnC,OAAO,EAAE,aAAa,EAAE,MAAM,aAAa,CAAC;AA2E5C;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA4EG;AACH,MAAM,OAAO,cAGX,SAAQ,WAAsE;IAC5E;;OAEG;IACH,YAAY,OAAkE,EAAE,MAAsB;QAClG,KAAK,CACD;YACI,GAAG,OAAO;YACV,sBAAsB,EAAE,GAAG,EAAE,CACzB,IAAI,CAAC,oBAAoB,EAAE;iBACtB,OAAO,CAAC;gBACL,MAAM,EAAE,KAAK,EAAE,OAAO,EAAE,EAAE,CAAC,MAAM,IAAI,CAAC,YAAY,CAAC,OAAO,CAAC;aAC9D,CAAC;iBACD,OAAO,CAAC,EAAE,MAAM,EAAE,KAAK,EAAE,OAAO,EAAE,EAAE,CAAC,MAAM,IAAI,CAAC,UAAU,CAAC,OAAO,CAAC,EAAE,CAAC;SAClF,EACD,MAAM,CACT,CAAC;IACN,CAAC;IAEO,KAAK,CAAC,YAAY,CAAC,eAA4C;QACnE,MAAM,KAAK,GAAG,eAAe,CAAC,WAAW,CAAC,IAAI,CAAC,QAAQ,CAAC,KAAK,CAAC,CAAC;QAC/D,MAAM,IAAI,GAAG,MAAM,CAAC,QAAQ,CAAC,eAAe,CAAC,IAAI,CAAC;YAC9C,CAAC,CAAC,eAAe,CAAC,IAAI,CAAC,QAAQ,CAAC,eAAe,CAAC,WAAW,CAAC,QAAQ,CAAC;YACrE,CAAC,CAAC,eAAe,CAAC,IAAI,CAAC;QAC3B,MAAM,GAAG,GAAG,aAAa,CAAC,IAAI,EAAE,EAAE,cAAc,EAAE,IAAI,EAAE,OAAO,EAAE,KAAK,EAAE,CAAC,CAAC;QAC1E,MAAM,CAAC,GAAG,OAAO,CAAC,IAAI,CAAC,GAAG,EAAE;YACxB,GAAG,EAAE,EAAE,cAAc,EAAE,IAAI,EAAE,OAAO,EAAE,KAAK,EAAE;SAC9B,CAAC,CAAC;QAErB,OAAO;YACH,CAAC;YACD,IAAI;SACP,CAAC;IACN,CAAC;IAEO,KAAK,CAAC,UAAU,CAAC,eAAgE;QACrF,MAAM,oBAAoB,GAAG,eAAe,CAAC,YAAY,CAAC;QAE1D,OAAO;YACH,YAAY,EAAE,KAAK,EAAE,cAAoC,EAAE,EAAE;gBACzD,OAAO,CAAC,MAAM,0BAA0B,CAAC;oBACrC,OAAO,EAAE,EAAE,GAAG,cAAc,EAAE,KAAK,EAAE,IAAI,CAAC,6BAA6B,CAAC,cAAc,EAAE,KAAK,CAAC,EAAE;oBAChG,CAAC,EAAE,eAAe,CAAC,CAAC;oBACpB,YAAY,EAAE,MAAM,IAAI,CAAC,eAAe,EAAE;oBAC1C,aAAa,EAAE,MAAM,IAAI,CAAC,sBAAsB,CAAC,eAAe,CAAC,OAAO,CAAC,GAAG,CAAC;oBAC7E,gBAAgB,EAAE,IAAI,CAAC,oBAAoB;oBAC3C,kBAAkB,EAAE,eAAe,CAAC,OAAO,CAAC,GAAG;oBAC/C,eAAe,EAAE,eAAe,CAAC,OAAO,CAAC,SAAS;oBAClD,YAAY,EAAE,oBAAoB;iBACrC,CAAC,CAA2B,CAAC,CAAC,2BAA2B;YAC9D,CAAC;YACD,eAAe,EAAE,KAAK,EAAE,QAAgB,EAAE,UAAmB,EAAE,EAAE;gBAC7D,IAAI,eAAe,CAAC,CAAC,CAAC,QAAQ,CAAC,CAAC,GAAG,EAAE,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;oBACjD,MAAM,IAAI,KAAK,CAAC,aAAa,QAAQ,cAAc,CAAC,CAAC;gBACzD,CAAC;YACL,CAAC;YACD,gBAAgB,EAAE,KAAK,EAAE,QAAiB,EAAE,SAAkB,EAAE,EAAE;gBAC9D,IAAI,QAAQ,EAAE,CAAC;oBACX,MAAM,eAAe,CAAC,eAAe,CAAC,QAAQ,EAAE,SAAS,CAAC,CAAC;gBAC/D,CAAC;gBAED,OAAO,eAAe,CAAC,CAAC,CAAC;YAC7B,CAAC;SACJ,CAAC;IACN,CAAC;CACJ;AAoBD,gBAAgB;AAChB,SAAS,oBAAoB,CACzB,OAAuE;IAEvE,OAAO,CAAC,CAAE,OAA4C,CAAC,YAAY,CAAC;AACxE,CAAC;AAED,gBAAgB;AAChB,MAAM,CAAC,KAAK,UAAU,0BAA0B,CAC5C,OAAuE;IAEvE,MAAM,EAAE,OAAO,EAAE,mBAAmB,EAAE,CAAC,EAAE,kBAAkB,EAAE,eAAe,EAAE,GAAG,OAAO,CAAC;IACzF,IAAI,CAAC,CAAC,EAAE,CAAC;QACL,MAAM,IAAI,KAAK,CAAC,wDAAwD,CAAC,CAAC;IAC9E,CAAC;IAED,MAAM,OAAO,GAAG,sCAAsC,CAAC;QACnD,eAAe,EAAE,mBAAmB,EAAE,QAAQ;QAC9C,eAAe;QACf,kBAAkB;QAClB,mBAAmB,EAAE,mBAAmB,EAAE,OAAO;KACpD,CAAC,CAAC;IAEH,MAAM,IAAI,GAAG,sBAAsB,CAC/B,CAAC,EACD,mBAAmB,EAAE,QAAQ,IAAI,GAAG,EACpC,mBAAmB,EAAE,OAAO,IAAI,eAAe,IAAI,kBAAkB,CACxE,CAAC;IAEF,IAAI,oBAAoB,CAAC,OAAO,CAAC,EAAE,CAAC;QAChC,OAAO,OAAO,CAAC,YAAY,CAAC;YACxB,IAAI;YACJ,OAAO;YACP,GAAG,mBAAmB;SACzB,CAAC,CAAC;IACP,CAAC;IACD,OAAO,YAAY,CAAC;QAChB,YAAY,EAAE,OAAO,CAAC,YAAY;QAClC,aAAa,EAAE,OAAO,CAAC,aAAa;QACpC,gBAAgB,EAAE,OAAO,CAAC,gBAAgB;QAC1C,IAAI;QACJ,OAAO;QACP,GAAG,mBAAmB;KACzB,CAAC,CAAC;AACP,CAAC;AAED;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,MAAM,UAAU,mBAAmB,CAGjC,MAAwC;IACtC,OAAO,MAAM,CAAC,MAAM,CAAU,MAAM,CAAC,CAAC;AAC1C,CAAC"}