@crawlee/cheerio 4.0.0-beta.121 → 4.0.0-beta.123

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,8 +1,7 @@
1
- import type { BasicCrawlingContext, CrawlingContext, EnqueueLinksOptions, ErrorHandler, GetUserDataFromRequest, HttpCrawlerOptions, InternalHttpCrawlingContext, InternalHttpHook, IRequestManager, RequestHandler, RouterHandler, RouterRoutes, RouteSchemas, RoutesFromSchemas, SkippedRequestCallback } from '@crawlee/http';
1
+ import type { AddRequestsBatchedResult, CrawlingContext, EnqueueLinksOptions, ErrorHandler, ExtractLinksOptions, GetUserDataFromRequest, HttpCrawlerOptions, InternalHttpCrawlingContext, InternalHttpHook, RequestHandler, RouterHandler, RouterRoutes, RouteSchemas, RoutesFromSchemas } from '@crawlee/http';
2
2
  import { HttpCrawler } from '@crawlee/http';
3
- import type { BatchAddRequestsResult, Dictionary } from '@crawlee/types';
3
+ import type { Dictionary } from '@crawlee/types';
4
4
  import { type CheerioRoot } from '@crawlee/utils/internal';
5
- import { type RobotsTxtFile } from '@crawlee/utils';
6
5
  import type { CheerioAPI } from 'cheerio';
7
6
  import * as cheerio from 'cheerio';
8
7
  export type CheerioErrorHandler<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
@@ -53,10 +52,14 @@ JSONData extends Dictionary = any> extends InternalHttpCrawlingContext<UserData,
53
52
  * ```
54
53
  */
55
54
  parseWithCheerio(selector?: string, timeoutMs?: number): Promise<CheerioRoot>;
55
+ /**
56
+ * Extracts URLs from the parsed HTML, without adding them to the request queue.
57
+ */
58
+ extractLinks(options?: ExtractLinksOptions): Promise<string[]>;
56
59
  /**
57
60
  * Helper function for extracting URLs from the parsed HTML and adding them to the request queue.
58
61
  */
59
- enqueueLinks(options?: EnqueueLinksOptions): Promise<BatchAddRequestsResult>;
62
+ enqueueLinks(options?: EnqueueLinksOptions): Promise<AddRequestsBatchedResult>;
60
63
  }
61
64
  export type CheerioRequestHandler<UserData extends Dictionary = any, // with default to Dictionary we cant use a typed router in untyped crawler
62
65
  JSONData extends Dictionary = any> = RequestHandler<CheerioCrawlingContext<UserData, JSONData>>;
@@ -149,31 +152,14 @@ export declare class CheerioCrawler<ContextExtension = Dictionary<never>, Extend
149
152
  readonly body: string;
150
153
  readonly $: CheerioAPI;
151
154
  } & {
152
- enqueueLinks: (enqueueOptions?: EnqueueLinksOptions) => Promise<BatchAddRequestsResult>;
155
+ extractLinks: (options?: ExtractLinksOptions) => Promise<string[]>;
156
+ enqueueLinks: (options?: EnqueueLinksOptions) => Promise<AddRequestsBatchedResult>;
153
157
  waitForSelector: (selector: string, _timeoutMs?: number) => Promise<void>;
154
158
  parseWithCheerio: (selector?: string, timeoutMs?: number) => Promise<CheerioAPI>;
155
159
  }>;
156
160
  private parseContent;
157
161
  private addHelpers;
158
162
  }
159
- interface EnqueueLinksInternalOptions {
160
- options?: EnqueueLinksOptions;
161
- $: cheerio.CheerioAPI | null;
162
- requestManager: IRequestManager;
163
- robotsTxtFile?: RobotsTxtFile;
164
- onSkippedRequest?: SkippedRequestCallback;
165
- originalRequestUrl: string;
166
- finalRequestUrl?: string;
167
- }
168
- interface BoundEnqueueLinksInternalOptions {
169
- enqueueLinks: BasicCrawlingContext['enqueueLinks'];
170
- options?: EnqueueLinksOptions;
171
- $: cheerio.CheerioAPI | null;
172
- originalRequestUrl: string;
173
- finalRequestUrl?: string;
174
- }
175
- /** @internal */
176
- export declare function cheerioCrawlerEnqueueLinks(options: EnqueueLinksInternalOptions | BoundEnqueueLinksInternalOptions): Promise<unknown>;
177
163
  /**
178
164
  * Creates new {@link Router} instance that works based on request labels.
179
165
  * This instance can then serve as a `requestHandler` of your {@link CheerioCrawler}.
@@ -201,4 +187,3 @@ export declare function cheerioCrawlerEnqueueLinks(options: EnqueueLinksInternal
201
187
  export declare function createCheerioRouter<Context extends CheerioCrawlingContext = CheerioCrawlingContext, Routes extends Record<keyof Routes, Dictionary> = Record<string, GetUserDataFromRequest<Context['request']>>>(routes?: RouterRoutes<Context, Routes>): RouterHandler<Context, Routes>;
202
188
  export declare function createCheerioRouter<Context extends CheerioCrawlingContext = CheerioCrawlingContext, UserData extends Dictionary = GetUserDataFromRequest<Context['request']>>(routes?: RouterRoutes<Context, Record<string, UserData>>): RouterHandler<Context, Record<string, UserData>>;
203
189
  export declare function createCheerioRouter<Context extends CheerioCrawlingContext = CheerioCrawlingContext, const Schemas extends RouteSchemas = RouteSchemas>(schemas: Schemas): RouterHandler<Context, RoutesFromSchemas<Schemas>>;
204
- export {};
@@ -1,4 +1,4 @@
1
- import { enqueueLinks, HttpCrawler, NavigationSkippedError, resolveBaseUrlForEnqueueLinksFiltering, Router, } from '@crawlee/http';
1
+ import { EnqueueStrategy, HttpCrawler, NavigationSkippedError, resolveBaseUrlForEnqueueLinksFiltering, Router, } from '@crawlee/http';
2
2
  import { extractUrlsFromCheerio } from '@crawlee/utils/internal';
3
3
  import * as cheerio from 'cheerio';
4
4
  import { parseDocument } from 'htmlparser2';
@@ -130,22 +130,28 @@ export class CheerioCrawler extends HttpCrawler {
130
130
  }
131
131
  }
132
132
  async addHelpers(crawlingContext) {
133
- const originalEnqueueLinks = crawlingContext.enqueueLinks;
133
+ const addRequests = crawlingContext.addRequests;
134
+ const extractLinks = async (options) => {
135
+ if (!crawlingContext.$) {
136
+ throw new Error('Cannot extract links because the DOM is not available.');
137
+ }
138
+ return extractUrlsFromCheerio(crawlingContext.$, options?.selector ?? 'a', options?.baseUrl ?? crawlingContext.request.loadedUrl ?? crawlingContext.request.url);
139
+ };
134
140
  return {
135
- enqueueLinks: async (enqueueOptions) => {
136
- return (await cheerioCrawlerEnqueueLinks({
137
- options: {
138
- ...enqueueOptions,
139
- limit: await this.calculateEnqueuedRequestLimit(enqueueOptions?.limit),
140
- },
141
- $: crawlingContext.$,
142
- requestManager: await this.getRequestManager(),
143
- robotsTxtFile: await this.getRobotsTxtFileForUrl(crawlingContext.request.url),
144
- onSkippedRequest: this.handleSkippedRequest,
145
- originalRequestUrl: crawlingContext.request.url,
141
+ extractLinks,
142
+ enqueueLinks: async (options = {}) => {
143
+ const baseUrl = resolveBaseUrlForEnqueueLinksFiltering({
144
+ enqueueStrategy: options.strategy,
146
145
  finalRequestUrl: crawlingContext.request.loadedUrl,
147
- enqueueLinks: originalEnqueueLinks,
148
- })); // TODO make this type safe, see https://github.com/apify/crawlee/issues/4024
146
+ originalRequestUrl: crawlingContext.request.url,
147
+ userProvidedBaseUrl: options.baseUrl,
148
+ });
149
+ const urls = await extractLinks(options);
150
+ return addRequests(urls, {
151
+ ...options,
152
+ baseUrl,
153
+ strategy: options.strategy ?? EnqueueStrategy.SameHostname,
154
+ });
149
155
  },
150
156
  waitForSelector: async (selector, _timeoutMs) => {
151
157
  if (crawlingContext.$(selector).get().length === 0) {
@@ -161,39 +167,6 @@ export class CheerioCrawler extends HttpCrawler {
161
167
  };
162
168
  }
163
169
  }
164
- /** @internal */
165
- function containsEnqueueLinks(options) {
166
- return !!options.enqueueLinks;
167
- }
168
- /** @internal */
169
- export async function cheerioCrawlerEnqueueLinks(options) {
170
- const { options: enqueueLinksOptions, $, originalRequestUrl, finalRequestUrl } = options;
171
- if (!$) {
172
- throw new Error('Cannot enqueue links because the DOM is not available.');
173
- }
174
- const baseUrl = resolveBaseUrlForEnqueueLinksFiltering({
175
- enqueueStrategy: enqueueLinksOptions?.strategy,
176
- finalRequestUrl,
177
- originalRequestUrl,
178
- userProvidedBaseUrl: enqueueLinksOptions?.baseUrl,
179
- });
180
- const urls = extractUrlsFromCheerio($, enqueueLinksOptions?.selector ?? 'a', enqueueLinksOptions?.baseUrl ?? finalRequestUrl ?? originalRequestUrl);
181
- if (containsEnqueueLinks(options)) {
182
- return options.enqueueLinks({
183
- urls,
184
- baseUrl,
185
- ...enqueueLinksOptions,
186
- });
187
- }
188
- return enqueueLinks({
189
- requestManager: options.requestManager,
190
- robotsTxtFile: options.robotsTxtFile,
191
- onSkippedRequest: options.onSkippedRequest,
192
- urls,
193
- baseUrl,
194
- ...enqueueLinksOptions,
195
- });
196
- }
197
170
  export function createCheerioRouter(routesOrSchemas) {
198
171
  return Router.create(routesOrSchemas);
199
172
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@crawlee/cheerio",
3
- "version": "4.0.0-beta.121",
3
+ "version": "4.0.0-beta.123",
4
4
  "description": "The scalable web crawling and scraping library for JavaScript/Node.js. Enables development of data extraction and web automation jobs (not only) with headless Chrome and Puppeteer.",
5
5
  "engines": {
6
6
  "node": ">=22.0.0"
@@ -47,9 +47,9 @@
47
47
  "access": "public"
48
48
  },
49
49
  "dependencies": {
50
- "@crawlee/http": "4.0.0-beta.121",
51
- "@crawlee/types": "4.0.0-beta.121",
52
- "@crawlee/utils": "4.0.0-beta.121",
50
+ "@crawlee/http": "4.0.0-beta.123",
51
+ "@crawlee/types": "4.0.0-beta.123",
52
+ "@crawlee/utils": "4.0.0-beta.123",
53
53
  "cheerio": "^1.0.0",
54
54
  "htmlparser2": "^10.0.0",
55
55
  "tslib": "^2.8.1"
@@ -61,5 +61,5 @@
61
61
  }
62
62
  }
63
63
  },
64
- "gitHead": "5027317de626f5ba6de5047ae9341a898258cc5a"
64
+ "gitHead": "f77648095c6a3f5ed8815c7620ea765db430ae44"
65
65
  }