@crawlee/cheerio 4.0.0-beta.167 → 4.0.0-beta.169
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/index.d.ts
CHANGED
|
@@ -1,8 +1,7 @@
|
|
|
1
|
-
import type {
|
|
2
|
-
import {
|
|
1
|
+
import type { CrawlingContext, DOMCrawlingContext, ErrorHandler, GetUserDataFromRequest, HttpCrawlerOptions, InternalHttpHook, RequestHandler, RouterHandler, RouterRoutes, RouteSchemas, RoutesFromSchemas } from '@crawlee/http';
|
|
2
|
+
import { DOMCrawler } from '@crawlee/http';
|
|
3
3
|
import type { Dictionary } from '@crawlee/types';
|
|
4
|
-
import type {
|
|
5
|
-
import * as cheerio from 'cheerio';
|
|
4
|
+
import type { CheerioParseResult } from './cheerio-parser.js';
|
|
6
5
|
export type CheerioErrorHandler<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
7
6
|
JSONData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
8
7
|
ContextExtension = Dictionary<never>> = ErrorHandler<CrawlingContext, CheerioCrawlingContext<UserData, JSONData> & ContextExtension>;
|
|
@@ -13,52 +12,7 @@ Routes extends Record<keyof Routes, Dictionary> = Record<string, UserData>, Stat
|
|
|
13
12
|
export type CheerioHook<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
14
13
|
JSONData extends Dictionary = any> = InternalHttpHook<CheerioCrawlingContext<UserData, JSONData>>;
|
|
15
14
|
export interface CheerioCrawlingContext<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
16
|
-
JSONData extends Dictionary = any> extends
|
|
17
|
-
/**
|
|
18
|
-
* The raw HTML content of the web page as a string.
|
|
19
|
-
*/
|
|
20
|
-
body: string;
|
|
21
|
-
/**
|
|
22
|
-
* The [Cheerio](https://cheerio.js.org/) object with parsed HTML.
|
|
23
|
-
* Cheerio is available only for HTML and XML content types.
|
|
24
|
-
*/
|
|
25
|
-
$: cheerio.CheerioAPI;
|
|
26
|
-
/**
|
|
27
|
-
* Wait for an element matching the selector to appear. Timeout is ignored.
|
|
28
|
-
*
|
|
29
|
-
* **Example usage:**
|
|
30
|
-
* ```ts
|
|
31
|
-
* async requestHandler({ waitForSelector, parseWithCheerio }) {
|
|
32
|
-
* await waitForSelector('article h1');
|
|
33
|
-
* const $ = await parseWithCheerio();
|
|
34
|
-
* const title = $('title').text();
|
|
35
|
-
* });
|
|
36
|
-
* ```
|
|
37
|
-
*/
|
|
38
|
-
waitForSelector(selector: string, timeoutMs?: number): Promise<void>;
|
|
39
|
-
/**
|
|
40
|
-
* Returns Cheerio handle, this is here to unify the crawler API, so they all have this handy method.
|
|
41
|
-
* It has the same return type as the `$` context property, use it only if you are abstracting your workflow to
|
|
42
|
-
* support different context types in one handler.
|
|
43
|
-
* When provided with the `selector` argument, it will throw if it's not available.
|
|
44
|
-
*
|
|
45
|
-
* **Example usage:**
|
|
46
|
-
* ```ts
|
|
47
|
-
* async requestHandler({ parseWithCheerio }) {
|
|
48
|
-
* const $ = await parseWithCheerio();
|
|
49
|
-
* const title = $('title').text();
|
|
50
|
-
* });
|
|
51
|
-
* ```
|
|
52
|
-
*/
|
|
53
|
-
parseWithCheerio(selector?: string, timeoutMs?: number): Promise<CheerioAPI>;
|
|
54
|
-
/**
|
|
55
|
-
* Extracts URLs from the parsed HTML, without adding them to the request queue.
|
|
56
|
-
*/
|
|
57
|
-
extractLinks(options?: ExtractLinksOptions): Promise<string[]>;
|
|
58
|
-
/**
|
|
59
|
-
* Helper function for extracting URLs from the parsed HTML and adding them to the request queue.
|
|
60
|
-
*/
|
|
61
|
-
enqueueLinks(options?: EnqueueLinksOptions): Promise<AddRequestsBatchedResult>;
|
|
15
|
+
JSONData extends Dictionary = any> extends DOMCrawlingContext<CheerioParseResult, UserData, JSONData> {
|
|
62
16
|
}
|
|
63
17
|
export type CheerioRequestHandler<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
|
|
64
18
|
JSONData extends Dictionary = any> = RequestHandler<CheerioCrawlingContext<UserData, JSONData>>;
|
|
@@ -141,14 +95,11 @@ JSONData extends Dictionary = any> = RequestHandler<CheerioCrawlingContext<UserD
|
|
|
141
95
|
* ```
|
|
142
96
|
* @category Crawlers
|
|
143
97
|
*/
|
|
144
|
-
export declare class CheerioCrawler<ContextExtension = Dictionary<never>, ExtendedContext extends CheerioCrawlingContext = CheerioCrawlingContext & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<CheerioCrawlingContext['request']>>, StatisticStateExtension extends object = {}> extends
|
|
98
|
+
export declare class CheerioCrawler<ContextExtension = Dictionary<never>, ExtendedContext extends CheerioCrawlingContext = CheerioCrawlingContext & ContextExtension, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<CheerioCrawlingContext['request']>>, StatisticStateExtension extends object = {}> extends DOMCrawler<CheerioParseResult, ContextExtension, ExtendedContext, Routes, StatisticStateExtension> {
|
|
145
99
|
/**
|
|
146
100
|
* All `CheerioCrawler` parameters are passed via an options object.
|
|
147
101
|
*/
|
|
148
102
|
constructor(options?: CheerioCrawlerOptions<ContextExtension, ExtendedContext, any, any, Routes, StatisticStateExtension>);
|
|
149
|
-
protected buildContextPipeline(): ContextPipeline<CrawlingContext, CheerioCrawlingContext>;
|
|
150
|
-
private parseContent;
|
|
151
|
-
private addHelpers;
|
|
152
103
|
}
|
|
153
104
|
/**
|
|
154
105
|
* Creates new {@link Router} instance that works based on request labels.
|
|
@@ -1,7 +1,5 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
3
|
-
import * as cheerio from 'cheerio';
|
|
4
|
-
import { parseDocument } from 'htmlparser2';
|
|
1
|
+
import { DOMCrawler, Router } from '@crawlee/http';
|
|
2
|
+
import { cheerioParser } from './cheerio-parser.js';
|
|
5
3
|
/**
|
|
6
4
|
* Provides a framework for the parallel crawling of web pages using plain HTTP requests and
|
|
7
5
|
* [cheerio](https://www.npmjs.com/package/cheerio) HTML parser.
|
|
@@ -81,90 +79,12 @@ import { parseDocument } from 'htmlparser2';
|
|
|
81
79
|
* ```
|
|
82
80
|
* @category Crawlers
|
|
83
81
|
*/
|
|
84
|
-
export class CheerioCrawler extends
|
|
82
|
+
export class CheerioCrawler extends DOMCrawler {
|
|
85
83
|
/**
|
|
86
84
|
* All `CheerioCrawler` parameters are passed via an options object.
|
|
87
85
|
*/
|
|
88
86
|
constructor(options) {
|
|
89
|
-
|
|
90
|
-
super({
|
|
91
|
-
...rest,
|
|
92
|
-
contextPipelineBuilder: contextPipelineBuilder ?? (() => this.buildContextPipeline()),
|
|
93
|
-
});
|
|
94
|
-
}
|
|
95
|
-
buildContextPipeline() {
|
|
96
|
-
return super
|
|
97
|
-
.buildContextPipeline()
|
|
98
|
-
.compose({
|
|
99
|
-
action: async (context) => await this.parseContent(context),
|
|
100
|
-
})
|
|
101
|
-
.compose({ action: async (context) => await this.addHelpers(context) });
|
|
102
|
-
}
|
|
103
|
-
async parseContent(crawlingContext) {
|
|
104
|
-
try {
|
|
105
|
-
const isXml = crawlingContext.contentType.type.includes('xml');
|
|
106
|
-
const body = Buffer.isBuffer(crawlingContext.body)
|
|
107
|
-
? crawlingContext.body.toString(crawlingContext.contentType.encoding)
|
|
108
|
-
: crawlingContext.body;
|
|
109
|
-
const dom = parseDocument(body, { decodeEntities: true, xmlMode: isXml });
|
|
110
|
-
const $ = cheerio.load(dom, {
|
|
111
|
-
xml: { decodeEntities: true, xmlMode: isXml },
|
|
112
|
-
});
|
|
113
|
-
return {
|
|
114
|
-
$,
|
|
115
|
-
body,
|
|
116
|
-
};
|
|
117
|
-
}
|
|
118
|
-
catch (err) {
|
|
119
|
-
if (err instanceof NavigationSkippedError) {
|
|
120
|
-
return {
|
|
121
|
-
get body() {
|
|
122
|
-
throw new NavigationSkippedError('The `body` property is not available - `skipNavigation` was used', { cause: err });
|
|
123
|
-
},
|
|
124
|
-
get $() {
|
|
125
|
-
throw new NavigationSkippedError('The `$` property is not available - `skipNavigation` was used', { cause: err });
|
|
126
|
-
},
|
|
127
|
-
};
|
|
128
|
-
}
|
|
129
|
-
throw err;
|
|
130
|
-
}
|
|
131
|
-
}
|
|
132
|
-
async addHelpers(crawlingContext) {
|
|
133
|
-
const addRequests = crawlingContext.addRequests;
|
|
134
|
-
const extractLinks = async (options) => {
|
|
135
|
-
if (!crawlingContext.$) {
|
|
136
|
-
throw new Error('Cannot extract links because the DOM is not available.');
|
|
137
|
-
}
|
|
138
|
-
return extractUrlsFromCheerio(crawlingContext.$, options?.selector ?? 'a', options?.baseUrl ?? crawlingContext.request.loadedUrl ?? crawlingContext.request.url);
|
|
139
|
-
};
|
|
140
|
-
return {
|
|
141
|
-
extractLinks,
|
|
142
|
-
enqueueLinks: async (options = {}) => {
|
|
143
|
-
const baseUrl = resolveBaseUrlForEnqueueLinksFiltering({
|
|
144
|
-
enqueueStrategy: options.strategy,
|
|
145
|
-
finalRequestUrl: crawlingContext.request.loadedUrl,
|
|
146
|
-
originalRequestUrl: crawlingContext.request.url,
|
|
147
|
-
userProvidedBaseUrl: options.baseUrl,
|
|
148
|
-
});
|
|
149
|
-
const urls = await extractLinks(options);
|
|
150
|
-
return addRequests(urls, {
|
|
151
|
-
...options,
|
|
152
|
-
baseUrl,
|
|
153
|
-
strategy: options.strategy ?? EnqueueStrategy.SameHostname,
|
|
154
|
-
});
|
|
155
|
-
},
|
|
156
|
-
waitForSelector: async (selector, _timeoutMs) => {
|
|
157
|
-
if (crawlingContext.$(selector).get().length === 0) {
|
|
158
|
-
throw new Error(`Selector '${selector}' not found.`);
|
|
159
|
-
}
|
|
160
|
-
},
|
|
161
|
-
parseWithCheerio: async (selector, timeoutMs) => {
|
|
162
|
-
if (selector) {
|
|
163
|
-
await crawlingContext.waitForSelector(selector, timeoutMs);
|
|
164
|
-
}
|
|
165
|
-
return crawlingContext.$;
|
|
166
|
-
},
|
|
167
|
-
};
|
|
87
|
+
super({ ...options, parser: cheerioParser() });
|
|
168
88
|
}
|
|
169
89
|
}
|
|
170
90
|
export function createCheerioRouter(routesOrSchemas) {
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import type { DOMParser } from '@crawlee/http';
|
|
2
|
+
import type { CheerioAPI } from 'cheerio';
|
|
3
|
+
export interface CheerioParseResult {
|
|
4
|
+
$: CheerioAPI;
|
|
5
|
+
body: string;
|
|
6
|
+
}
|
|
7
|
+
/**
|
|
8
|
+
* A {@link DOMParser} backed by [cheerio](https://www.npmjs.com/package/cheerio). Pass it to a
|
|
9
|
+
* {@link DOMCrawler} to get the crawling context {@link CheerioCrawler} provides.
|
|
10
|
+
*/
|
|
11
|
+
export declare function cheerioParser(): DOMParser<CheerioParseResult>;
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
import { extractUrlsFromCheerio } from '@crawlee/utils/internal';
|
|
2
|
+
import * as cheerio from 'cheerio';
|
|
3
|
+
import { parseDocument } from 'htmlparser2';
|
|
4
|
+
/**
|
|
5
|
+
* A {@link DOMParser} backed by [cheerio](https://www.npmjs.com/package/cheerio). Pass it to a
|
|
6
|
+
* {@link DOMCrawler} to get the crawling context {@link CheerioCrawler} provides.
|
|
7
|
+
*/
|
|
8
|
+
export function cheerioParser() {
|
|
9
|
+
return {
|
|
10
|
+
placeholderMembers: { $: true, body: true },
|
|
11
|
+
parse(context) {
|
|
12
|
+
const isXml = context.contentType.type.includes('xml');
|
|
13
|
+
const body = Buffer.isBuffer(context.body)
|
|
14
|
+
? context.body.toString(context.contentType.encoding)
|
|
15
|
+
: context.body;
|
|
16
|
+
const dom = parseDocument(body, { decodeEntities: true, xmlMode: isXml });
|
|
17
|
+
const $ = cheerio.load(dom, {
|
|
18
|
+
xml: { decodeEntities: true, xmlMode: isXml },
|
|
19
|
+
});
|
|
20
|
+
return { $, body };
|
|
21
|
+
},
|
|
22
|
+
extractLinks: ({ $ }, selector, baseUrl) => extractUrlsFromCheerio($, selector, baseUrl),
|
|
23
|
+
select: ({ $ }, selector) => $(selector).get(),
|
|
24
|
+
toCheerio: ({ $ }) => $,
|
|
25
|
+
};
|
|
26
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@crawlee/cheerio",
|
|
3
|
-
"version": "4.0.0-beta.
|
|
3
|
+
"version": "4.0.0-beta.169",
|
|
4
4
|
"description": "The scalable web crawling and scraping library for JavaScript/Node.js. Enables development of data extraction and web automation jobs (not only) with headless Chrome and Puppeteer.",
|
|
5
5
|
"engines": {
|
|
6
6
|
"node": ">=22.0.0"
|
|
@@ -47,9 +47,9 @@
|
|
|
47
47
|
"access": "public"
|
|
48
48
|
},
|
|
49
49
|
"dependencies": {
|
|
50
|
-
"@crawlee/http": "4.0.0-beta.
|
|
51
|
-
"@crawlee/types": "4.0.0-beta.
|
|
52
|
-
"@crawlee/utils": "4.0.0-beta.
|
|
50
|
+
"@crawlee/http": "4.0.0-beta.169",
|
|
51
|
+
"@crawlee/types": "4.0.0-beta.169",
|
|
52
|
+
"@crawlee/utils": "4.0.0-beta.169",
|
|
53
53
|
"cheerio": "^1.0.0",
|
|
54
54
|
"htmlparser2": "^10.0.0",
|
|
55
55
|
"tslib": "^2.8.1"
|
|
@@ -61,5 +61,5 @@
|
|
|
61
61
|
}
|
|
62
62
|
}
|
|
63
63
|
},
|
|
64
|
-
"gitHead": "
|
|
64
|
+
"gitHead": "a42dc93fa0ef68180d83ab12af362082690f73e2"
|
|
65
65
|
}
|