@crawlee/http 4.0.0-beta.187 → 4.0.0-beta.189

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -226,6 +226,7 @@ export class HttpCrawler extends BasicCrawler {
226
226
  request.state = RequestState.BEFORE_NAV;
227
227
  return {};
228
228
  }
229
+ // oxlint-disable-next-line crawlee/prefer-private-fields -- patched by @crawlee/otel
229
230
  async makeHttpRequest(crawlingContext) {
230
231
  tryCancel();
231
232
  const { request, session } = crawlingContext;
@@ -382,13 +383,12 @@ export class HttpCrawler extends BasicCrawler {
382
383
  async parseResponse(request, response) {
383
384
  const { status } = response;
384
385
  const { type, charset } = parseContentTypeFromResponse(response);
385
- const { response: reencodedResponse, encoding } = this.encodeResponse(request, response, charset);
386
- const contentType = { type, encoding };
387
386
  if (status >= 400 && status <= 599) {
388
387
  this.statistics.registerStatusCode(status);
389
388
  }
390
389
  if (this.isErrorStatusCode(status)) {
391
- const body = await reencodedResponse.text(); // TODO - this always uses UTF-8 (see https://developer.mozilla.org/en-US/docs/Web/API/Request/text)
390
+ const { response: decoded } = this.encodeResponse(request, response, charset);
391
+ const body = await decoded.text(); // TODO - this always uses UTF-8 (see https://developer.mozilla.org/en-US/docs/Web/API/Request/text)
392
392
  // Errors are often sent as JSON, so attempt to parse them,
393
393
  // despite Accept header being set to text/html.
394
394
  if (type === APPLICATION_JSON_MIME_TYPE) {
@@ -404,26 +404,27 @@ export class HttpCrawler extends BasicCrawler {
404
404
  // It's not a JSON, so it's probably some text. Get the first 100 chars of it.
405
405
  throw new Error(`${status} - Internal Server Error: ${body.slice(0, 100)}`);
406
406
  }
407
- else if (HTML_AND_XML_MIME_TYPES.includes(type)) {
408
- if (!charset && !this.#forceResponseEncoding) {
409
- const rawBytes = Buffer.from(await response.arrayBuffer());
410
- const metaCharset = extractCharsetFromHtmlBytes(rawBytes);
411
- const charsetToUse = metaCharset ?? this.#suggestResponseEncoding ?? 'utf-8';
412
- const body = iconv.encodingExists(charsetToUse)
413
- ? iconv.decode(rawBytes, charsetToUse)
414
- : rawBytes.toString('utf8');
415
- return { response, contentType: { type, encoding: 'utf-8' }, body };
416
- }
417
- return { response, contentType, body: await reencodedResponse.text() };
407
+ if (HTML_AND_XML_MIME_TYPES.includes(type) && !charset && !this.#forceResponseEncoding) {
408
+ // The charset comes from the document itself, so the raw bytes are what we need -
409
+ // decoding them through `encodeResponse` first would consume the body for nothing.
410
+ const rawBytes = Buffer.from(await response.arrayBuffer());
411
+ const metaCharset = extractCharsetFromHtmlBytes(rawBytes);
412
+ const charsetToUse = metaCharset ?? this.#suggestResponseEncoding ?? 'utf-8';
413
+ const body = iconv.encodingExists(charsetToUse)
414
+ ? iconv.decode(rawBytes, charsetToUse)
415
+ : rawBytes.toString('utf8');
416
+ return { response, contentType: { type, encoding: 'utf-8' }, body };
418
417
  }
419
- else {
420
- const body = Buffer.from(await reencodedResponse.bytes());
421
- return {
422
- body,
423
- response,
424
- contentType,
425
- };
418
+ const { response: reencodedResponse, encoding } = this.encodeResponse(request, response, charset);
419
+ const contentType = { type, encoding };
420
+ if (HTML_AND_XML_MIME_TYPES.includes(type)) {
421
+ return { response, contentType, body: await reencodedResponse.text() };
426
422
  }
423
+ return {
424
+ body: Buffer.from(await reencodedResponse.bytes()),
425
+ response,
426
+ contentType,
427
+ };
427
428
  }
428
429
  /**
429
430
  * Combines the provided `requestOptions` with mandatory (non-overridable) values.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@crawlee/http",
3
- "version": "4.0.0-beta.187",
3
+ "version": "4.0.0-beta.189",
4
4
  "description": "The scalable web crawling and scraping library for JavaScript/Node.js. Enables development of data extraction and web automation jobs (not only) with headless Chrome and Puppeteer.",
5
5
  "engines": {
6
6
  "node": ">=22.0.0"
@@ -49,11 +49,11 @@
49
49
  "dependencies": {
50
50
  "@apify/timeout": "^1.0.1",
51
51
  "@apify/utilities": "^3.0.1",
52
- "@crawlee/basic": "4.0.0-beta.187",
53
- "@crawlee/core": "4.0.0-beta.187",
54
- "@crawlee/http-client": "4.0.0-beta.187",
55
- "@crawlee/types": "4.0.0-beta.187",
56
- "@crawlee/utils": "4.0.0-beta.187",
52
+ "@crawlee/basic": "4.0.0-beta.189",
53
+ "@crawlee/core": "4.0.0-beta.189",
54
+ "@crawlee/http-client": "4.0.0-beta.189",
55
+ "@crawlee/types": "4.0.0-beta.189",
56
+ "@crawlee/utils": "4.0.0-beta.189",
57
57
  "@types/content-type": "^1.1.8",
58
58
  "cheerio": "^1.0.0",
59
59
  "content-type": "^1.0.5",
@@ -70,5 +70,5 @@
70
70
  }
71
71
  }
72
72
  },
73
- "gitHead": "1281760f20f7ee1181e3171d759efb055647e6c5"
73
+ "gitHead": "1dbb0f272b9da9c7a6bf3bdc1ae48f84eda3f21c"
74
74
  }