@crawlee/http 4.0.0-beta.175 → 4.0.0-beta.177

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -86,7 +86,11 @@ export class DOMCrawler extends HttpCrawler {
86
86
  });
87
87
  },
88
88
  async parseWithCheerio(selector, _timeoutMs = 5_000) {
89
- const $ = (await parser.toCheerio?.(context)) ?? (await import('cheerio')).load(context.body);
89
+ // Import full cheerio (not cheerio/slim) to be browser-compliant for DOM Crawlers (not CheerioCrawler).
90
+ const $ = (await parser.toCheerio?.(context)) ??
91
+ (await import('cheerio')).load(context.body, {
92
+ xmlMode: context.contentType.type.includes('xml'),
93
+ });
90
94
  if (selector && $(selector).get().length === 0) {
91
95
  throw new Error(`Selector '${selector}' not found.`);
92
96
  }
@@ -286,16 +286,18 @@ export class HttpCrawler extends BasicCrawler {
286
286
  tryCancel();
287
287
  const response = parsed.response;
288
288
  const contentType = parsed.contentType;
289
+ const loadBody = async () => {
290
+ const { load } = await import('cheerio/slim');
291
+ return load(parsed.body.toString(), { xmlMode: contentType.type.includes('xml') });
292
+ };
289
293
  const waitForSelector = async (selector, _timeoutMs) => {
290
- const cheerio = await import('cheerio');
291
- const $ = cheerio.load(parsed.body.toString());
294
+ const $ = await loadBody();
292
295
  if ($(selector).get().length === 0) {
293
296
  throw new Error(`Selector '${selector}' not found.`);
294
297
  }
295
298
  };
296
299
  const parseWithCheerio = async (selector, timeoutMs) => {
297
- const cheerio = await import('cheerio');
298
- const $ = cheerio.load(parsed.body.toString());
300
+ const $ = await loadBody();
299
301
  if (selector) {
300
302
  await crawlingContext.waitForSelector(selector, timeoutMs);
301
303
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@crawlee/http",
3
- "version": "4.0.0-beta.175",
3
+ "version": "4.0.0-beta.177",
4
4
  "description": "The scalable web crawling and scraping library for JavaScript/Node.js. Enables development of data extraction and web automation jobs (not only) with headless Chrome and Puppeteer.",
5
5
  "engines": {
6
6
  "node": ">=22.0.0"
@@ -49,11 +49,11 @@
49
49
  "dependencies": {
50
50
  "@apify/timeout": "^1.0.1",
51
51
  "@apify/utilities": "^3.0.1",
52
- "@crawlee/basic": "4.0.0-beta.175",
53
- "@crawlee/core": "4.0.0-beta.175",
54
- "@crawlee/http-client": "4.0.0-beta.175",
55
- "@crawlee/types": "4.0.0-beta.175",
56
- "@crawlee/utils": "4.0.0-beta.175",
52
+ "@crawlee/basic": "4.0.0-beta.177",
53
+ "@crawlee/core": "4.0.0-beta.177",
54
+ "@crawlee/http-client": "4.0.0-beta.177",
55
+ "@crawlee/types": "4.0.0-beta.177",
56
+ "@crawlee/utils": "4.0.0-beta.177",
57
57
  "@types/content-type": "^1.1.8",
58
58
  "cheerio": "^1.0.0",
59
59
  "content-type": "^1.0.5",
@@ -70,5 +70,5 @@
70
70
  }
71
71
  }
72
72
  },
73
- "gitHead": "9ea1e6e846bc03c6623a0311c7f36976b1c5bae9"
73
+ "gitHead": "bfd191ba71902fc682dd21fb9ab17590037a60d5"
74
74
  }