@crawlee/utils 4.0.0-beta.14 → 4.0.0-beta.140

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/README.md +14 -14
  2. package/index.d.ts +6 -15
  3. package/index.js +5 -14
  4. package/internal.d.ts +9 -0
  5. package/internal.js +8 -0
  6. package/internals/blocked.d.ts +0 -1
  7. package/internals/blocked.js +0 -1
  8. package/internals/cheerio.d.ts +1 -3
  9. package/internals/cheerio.js +4 -5
  10. package/internals/extract-urls.d.ts +5 -1
  11. package/internals/extract-urls.js +23 -18
  12. package/internals/general.d.ts +0 -25
  13. package/internals/general.js +2 -110
  14. package/internals/iterables.d.ts +9 -70
  15. package/internals/iterables.js +40 -111
  16. package/internals/open_graph_parser.d.ts +2 -3
  17. package/internals/open_graph_parser.js +8 -9
  18. package/internals/robots.d.ts +38 -12
  19. package/internals/robots.js +72 -48
  20. package/internals/schemas.d.ts +114 -0
  21. package/internals/schemas.js +114 -0
  22. package/internals/sitemap.d.ts +69 -8
  23. package/internals/sitemap.js +239 -75
  24. package/internals/social.d.ts +1 -2
  25. package/internals/social.js +7 -5
  26. package/internals/url.d.ts +70 -2
  27. package/internals/url.js +120 -2
  28. package/internals/validation.d.ts +25 -0
  29. package/internals/validation.js +140 -0
  30. package/package.json +9 -8
  31. package/index.d.ts.map +0 -1
  32. package/index.js.map +0 -1
  33. package/internals/blocked.d.ts.map +0 -1
  34. package/internals/blocked.js.map +0 -1
  35. package/internals/cheerio.d.ts.map +0 -1
  36. package/internals/cheerio.js.map +0 -1
  37. package/internals/chunk.d.ts +0 -2
  38. package/internals/chunk.d.ts.map +0 -1
  39. package/internals/chunk.js +0 -40
  40. package/internals/chunk.js.map +0 -1
  41. package/internals/debug.d.ts +0 -32
  42. package/internals/debug.d.ts.map +0 -1
  43. package/internals/debug.js +0 -41
  44. package/internals/debug.js.map +0 -1
  45. package/internals/extract-urls.d.ts.map +0 -1
  46. package/internals/extract-urls.js.map +0 -1
  47. package/internals/general.d.ts.map +0 -1
  48. package/internals/general.js.map +0 -1
  49. package/internals/iterables.d.ts.map +0 -1
  50. package/internals/iterables.js.map +0 -1
  51. package/internals/open_graph_parser.d.ts.map +0 -1
  52. package/internals/open_graph_parser.js.map +0 -1
  53. package/internals/robots.d.ts.map +0 -1
  54. package/internals/robots.js.map +0 -1
  55. package/internals/sitemap.d.ts.map +0 -1
  56. package/internals/sitemap.js.map +0 -1
  57. package/internals/social.d.ts.map +0 -1
  58. package/internals/social.js.map +0 -1
  59. package/internals/system-info/cpu-info.d.ts +0 -64
  60. package/internals/system-info/cpu-info.d.ts.map +0 -1
  61. package/internals/system-info/cpu-info.js +0 -211
  62. package/internals/system-info/cpu-info.js.map +0 -1
  63. package/internals/system-info/memory-info.d.ts +0 -28
  64. package/internals/system-info/memory-info.d.ts.map +0 -1
  65. package/internals/system-info/memory-info.js +0 -118
  66. package/internals/system-info/memory-info.js.map +0 -1
  67. package/internals/system-info/ps-tree.d.ts +0 -18
  68. package/internals/system-info/ps-tree.d.ts.map +0 -1
  69. package/internals/system-info/ps-tree.js +0 -145
  70. package/internals/system-info/ps-tree.js.map +0 -1
  71. package/internals/typedefs.d.ts +0 -5
  72. package/internals/typedefs.d.ts.map +0 -1
  73. package/internals/typedefs.js +0 -9
  74. package/internals/typedefs.js.map +0 -1
  75. package/internals/url.d.ts.map +0 -1
  76. package/internals/url.js.map +0 -1
@@ -2,29 +2,31 @@ import { createHash } from 'node:crypto';
2
2
  import { PassThrough, pipeline, Readable, Transform } from 'node:stream';
3
3
  import { StringDecoder } from 'node:string_decoder';
4
4
  import { createGunzip } from 'node:zlib';
5
- import sax from 'sax';
5
+ import { FetchHttpClient } from '@crawlee/http-client';
6
6
  import MIMEType from 'whatwg-mimetype';
7
- import log from '@apify/log';
7
+ import { mergeAsyncIterables } from './iterables.js';
8
+ import { RobotsTxtFile } from './robots.js';
9
+ import { filterUrl } from './url.js';
8
10
  class SitemapTxtParser extends Transform {
9
- decoder = new StringDecoder('utf8');
10
- buffer = '';
11
+ #decoder = new StringDecoder('utf8');
12
+ #buffer = '';
11
13
  constructor() {
12
14
  super({
13
15
  readableObjectMode: true,
14
16
  transform: (chunk, _encoding, callback) => {
15
- this.processBuffer(this.decoder.write(chunk), false);
17
+ this.processBuffer(this.#decoder.write(chunk), false);
16
18
  callback();
17
19
  },
18
20
  flush: (callback) => {
19
- this.processBuffer(this.decoder.end(), true);
21
+ this.processBuffer(this.#decoder.end(), true);
20
22
  callback();
21
23
  },
22
24
  });
23
25
  }
24
26
  processBuffer(input, finalize) {
25
- this.buffer += input;
26
- if (finalize || this.buffer.includes('\n')) {
27
- const parts = this.buffer
27
+ this.#buffer += input;
28
+ if (finalize || this.#buffer.includes('\n')) {
29
+ const parts = this.#buffer
28
30
  .split('\n')
29
31
  .map((part) => part.trim())
30
32
  .filter((part) => part.length > 0);
@@ -32,101 +34,109 @@ class SitemapTxtParser extends Transform {
32
34
  for (const url of parts) {
33
35
  this.push({ type: 'url', loc: url });
34
36
  }
35
- this.buffer = '';
37
+ this.#buffer = '';
36
38
  }
37
39
  else if (parts.length > 0) {
38
40
  for (const url of parts.slice(0, -1)) {
39
41
  this.push({ type: 'url', loc: url });
40
42
  }
41
- this.buffer = parts.at(-1);
43
+ this.#buffer = parts.at(-1);
42
44
  }
43
45
  }
44
46
  }
45
47
  }
46
48
  class SitemapXmlParser extends Transform {
47
- decoder = new StringDecoder('utf8');
48
- parser = new sax.SAXParser(true);
49
- rootTagName;
50
- currentTag = undefined;
51
- url = {};
52
- constructor() {
49
+ #decoder = new StringDecoder('utf8');
50
+ #parser;
51
+ #rootTagName;
52
+ #currentTag = undefined;
53
+ #url = {};
54
+ static async create() {
55
+ const { SAXParser } = await import('sax');
56
+ return new SitemapXmlParser(new SAXParser(true));
57
+ }
58
+ constructor(parser) {
53
59
  super({
54
60
  readableObjectMode: true,
55
61
  transform: (chunk, _encoding, callback) => {
56
- this.parser.write(this.decoder.write(chunk));
62
+ this.#parser.write(this.#decoder.write(chunk));
57
63
  callback();
58
64
  },
59
65
  flush: (callback) => {
60
- const rest = this.decoder.end();
66
+ const rest = this.#decoder.end();
61
67
  if (rest.length > 0) {
62
- this.parser.write(rest);
68
+ this.#parser.write(rest);
63
69
  }
64
- this.parser.end();
70
+ this.#parser.end();
65
71
  callback();
66
72
  },
67
73
  });
68
- this.parser.onopentag = this.onOpenTag.bind(this);
69
- this.parser.onclosetag = this.onCloseTag.bind(this);
70
- this.parser.ontext = this.onText.bind(this);
71
- this.parser.oncdata = this.onText.bind(this);
72
- this.parser.onerror = this.destroy.bind(this);
74
+ this.#parser = parser;
75
+ this.#parser.onopentag = this.onOpenTag.bind(this);
76
+ this.#parser.onclosetag = this.onCloseTag.bind(this);
77
+ this.#parser.ontext = this.onText.bind(this);
78
+ this.#parser.oncdata = this.onText.bind(this);
79
+ this.#parser.onerror = this.destroy.bind(this);
73
80
  }
74
81
  onOpenTag(node) {
75
- if (this.rootTagName !== undefined) {
82
+ if (this.#rootTagName !== undefined) {
76
83
  if (node.name === 'loc' ||
77
84
  node.name === 'lastmod' ||
78
85
  node.name === 'priority' ||
79
86
  node.name === 'changefreq') {
80
- this.currentTag = node.name;
87
+ this.#currentTag = node.name;
81
88
  }
82
89
  }
83
90
  if (node.name === 'urlset') {
84
- this.rootTagName = 'urlset';
91
+ this.#rootTagName = 'urlset';
85
92
  }
86
93
  if (node.name === 'sitemapindex') {
87
- this.rootTagName = 'sitemapindex';
94
+ this.#rootTagName = 'sitemapindex';
88
95
  }
89
96
  }
90
97
  onCloseTag(name) {
91
98
  if (name === 'loc' || name === 'lastmod' || name === 'priority' || name === 'changefreq') {
92
- this.currentTag = undefined;
99
+ this.#currentTag = undefined;
93
100
  }
94
- if (name === 'url' && this.url.loc !== undefined) {
95
- this.push({ type: 'url', ...this.url, loc: this.url.loc });
96
- this.url = {};
101
+ if (name === 'url') {
102
+ if (this.#url.loc !== undefined) {
103
+ this.push({ type: 'url', ...this.#url, loc: this.#url.loc });
104
+ }
105
+ this.#url = {};
97
106
  }
98
107
  }
99
108
  onText(text) {
100
- if (this.currentTag === 'loc') {
101
- if (this.rootTagName === 'sitemapindex') {
109
+ if (this.#currentTag === 'loc') {
110
+ if (this.#rootTagName === 'sitemapindex') {
102
111
  this.push({ type: 'sitemapUrl', url: text.trim() });
103
112
  }
104
- if (this.rootTagName === 'urlset') {
105
- this.url ??= {};
106
- this.url.loc = text.trim();
113
+ if (this.#rootTagName === 'urlset') {
114
+ this.#url ??= {};
115
+ this.#url.loc = text.trim();
107
116
  }
108
117
  }
109
118
  text = text.trim();
110
- if (this.currentTag === 'lastmod') {
111
- this.url.lastmod = new Date(text);
119
+ if (this.#currentTag === 'lastmod') {
120
+ const lastmod = new Date(text);
121
+ if (!Number.isNaN(lastmod.getTime())) {
122
+ this.#url.lastmod = lastmod;
123
+ }
112
124
  }
113
- if (this.currentTag === 'priority') {
114
- this.url.priority = Number(text);
125
+ if (this.#currentTag === 'priority') {
126
+ this.#url.priority = Number(text);
115
127
  }
116
- if (this.currentTag === 'changefreq') {
128
+ if (this.#currentTag === 'changefreq') {
117
129
  if (['always', 'hourly', 'daily', 'weekly', 'monthly', 'yearly', 'never'].includes(text)) {
118
- this.url.changefreq = text;
130
+ this.#url.changefreq = text;
119
131
  }
120
132
  }
121
133
  }
122
134
  }
123
135
  export async function* parseSitemap(initialSources, proxyUrl, options) {
124
- const { gotScraping } = await import('got-scraping');
125
- const { fileTypeStream } = await import('file-type');
126
- const { emitNestedSitemaps = false, maxDepth = Infinity, sitemapRetries = 3, networkTimeouts, reportNetworkErrors = true, } = options ?? {};
136
+ const { httpClient = new FetchHttpClient(), emitNestedSitemaps = false, maxDepth = Infinity, sitemapRetries = 3, timeoutMillis: timeout = 30000, reportNetworkErrors = true, nestedSitemapFilter, enqueueStrategy = 'same-hostname', logger, } = options ?? {};
127
137
  const sources = [...initialSources];
128
138
  const visitedSitemapUrls = new Set();
129
- const createParser = (contentType = '', url) => {
139
+ const createParser = async (contentType = '', url) => {
130
140
  let mimeType;
131
141
  try {
132
142
  mimeType = new MIMEType(contentType);
@@ -135,7 +145,7 @@ export async function* parseSitemap(initialSources, proxyUrl, options) {
135
145
  mimeType = null;
136
146
  }
137
147
  if (mimeType?.isXML() || url?.pathname.endsWith('.xml')) {
138
- return new SitemapXmlParser();
148
+ return SitemapXmlParser.create();
139
149
  }
140
150
  if (mimeType?.essence === 'text/plain' || url?.pathname.endsWith('.txt')) {
141
151
  return new SitemapTxtParser();
@@ -145,33 +155,40 @@ export async function* parseSitemap(initialSources, proxyUrl, options) {
145
155
  while (sources.length > 0) {
146
156
  const source = sources.shift();
147
157
  if ((source?.depth ?? 0) > maxDepth) {
148
- log.debug(`Skipping sitemap ${source.type === 'url' ? source.url : ''} because it reached max depth ${maxDepth}.`);
149
158
  continue;
150
159
  }
151
160
  let items = null;
161
+ // Parent URL, parsed once and reused as the origin for the strategy checks below.
162
+ let sitemapUrl;
152
163
  if (source.type === 'url') {
153
- const sitemapUrl = new URL(source.url);
164
+ sitemapUrl = new URL(source.url);
154
165
  visitedSitemapUrls.add(sitemapUrl.toString());
155
166
  let retriesLeft = sitemapRetries + 1;
156
167
  while (retriesLeft-- > 0) {
157
168
  try {
158
- const sitemapStream = await new Promise((resolve, reject) => {
159
- const request = gotScraping.stream({
160
- url: sitemapUrl,
161
- proxyUrl,
169
+ let sitemapResponse;
170
+ try {
171
+ sitemapResponse = await httpClient.sendRequest(new Request(sitemapUrl, {
162
172
  method: 'GET',
163
- timeout: networkTimeouts,
164
173
  headers: {
165
- accept: 'text/plain, application/xhtml+xml, application/xml;q=0.9, */*;q=0.8',
174
+ accept: '*/*',
166
175
  },
176
+ }), {
177
+ proxyUrl,
178
+ timeoutMillis: timeout,
167
179
  });
168
- request.on('response', () => resolve(request));
169
- request.on('error', reject);
170
- });
180
+ }
181
+ catch (error) {
182
+ sitemapResponse = null;
183
+ }
171
184
  let error = null;
172
- if (sitemapStream.response.statusCode >= 200 && sitemapStream.response.statusCode < 300) {
173
- let contentType = sitemapStream.response.headers['content-type'];
174
- const streamWithType = await fileTypeStream(sitemapStream);
185
+ if (sitemapResponse && sitemapResponse.status >= 200 && sitemapResponse.status < 300) {
186
+ let contentType = sitemapResponse.headers.get('content-type');
187
+ if (sitemapResponse.body === null) {
188
+ break;
189
+ }
190
+ const { fileTypeStream } = await import('file-type');
191
+ const streamWithType = await fileTypeStream(Readable.fromWeb(sitemapResponse.body));
175
192
  if (streamWithType.fileType !== undefined) {
176
193
  contentType = streamWithType.fileType.mime;
177
194
  }
@@ -184,7 +201,7 @@ export async function* parseSitemap(initialSources, proxyUrl, options) {
184
201
  sitemapUrl.pathname = sitemapUrl.pathname.substring(0, sitemapUrl.pathname.length - 3);
185
202
  }
186
203
  }
187
- items = pipeline(streamWithType, isGzipped ? createGunzip() : new PassThrough(), createParser(contentType, sitemapUrl), (e) => {
204
+ items = pipeline(streamWithType, isGzipped ? createGunzip() : new PassThrough(), await createParser(contentType ?? undefined, sitemapUrl), (e) => {
188
205
  if (e !== undefined && e !== null) {
189
206
  error = { type: 'parser', error: e };
190
207
  }
@@ -193,7 +210,7 @@ export async function* parseSitemap(initialSources, proxyUrl, options) {
193
210
  else {
194
211
  error = {
195
212
  type: 'fetch',
196
- error: new Error(`Failed to fetch sitemap: ${sitemapUrl}, status code: ${sitemapStream.response.statusCode}`),
213
+ error: new Error(`Failed to fetch sitemap: ${sitemapUrl}, status code: ${sitemapResponse?.status}`),
197
214
  };
198
215
  }
199
216
  if (error !== null) {
@@ -207,28 +224,53 @@ export async function* parseSitemap(initialSources, proxyUrl, options) {
207
224
  }
208
225
  }
209
226
  catch (e) {
210
- log.warning(`Malformed sitemap content: ${sitemapUrl}, ${retriesLeft === 0 ? 'no retries left.' : 'retrying...'} (${e})`);
227
+ logger?.warning(`Malformed sitemap content: ${sitemapUrl}, ${retriesLeft === 0 ? 'no retries left.' : 'retrying...'} (${e})`);
211
228
  }
212
229
  }
213
230
  }
214
231
  else if (source.type === 'raw') {
215
- items = pipeline(Readable.from([source.content]), createParser('text/xml'), (error) => {
232
+ items = pipeline(Readable.from([source.content]), await createParser('text/xml'), (error) => {
216
233
  if (error !== undefined) {
217
- log.warning(`Malformed sitemap content: ${error}`);
234
+ logger?.warning(`Malformed sitemap content: ${error}`);
218
235
  }
219
236
  });
220
237
  }
221
238
  if (items === null) {
222
239
  continue;
223
240
  }
241
+ // URL entries dropped by the enqueue strategy filter, reported in one warning per sitemap after
242
+ // the loop (per-entry warnings could flood the log; individual drops are logged at debug level).
243
+ let droppedUrlEntries = 0;
224
244
  for await (const item of items) {
225
245
  if (item.type === 'sitemapUrl' && !visitedSitemapUrls.has(item.url)) {
246
+ if (nestedSitemapFilter && !nestedSitemapFilter(item.url)) {
247
+ logger?.debug(`Skipping sitemap ${item.url} due to nestedSitemapFilter.`);
248
+ continue;
249
+ }
250
+ // Keep only nested sitemaps matching the strategy (and using http(s)) relative to the
251
+ // parent. Raw string sources have no parent URL, so the check is skipped.
252
+ if (source.type === 'url') {
253
+ const { allowed, reason } = filterUrl(item.url, sitemapUrl, enqueueStrategy);
254
+ if (!allowed) {
255
+ logger?.warning(`Skipping nested sitemap ${item.url} (parent ${source.url}): ${reason}.`);
256
+ continue;
257
+ }
258
+ }
226
259
  sources.push({ type: 'url', url: item.url, depth: (source.depth ?? 0) + 1 });
227
260
  if (emitNestedSitemaps) {
228
261
  yield { loc: item.url, originSitemapUrl: null };
229
262
  }
230
263
  }
231
264
  if (item.type === 'url') {
265
+ // Keep only URL entries that match the enqueue strategy relative to the parent (see above).
266
+ if (source.type === 'url') {
267
+ const { allowed, reason } = filterUrl(item.loc, sitemapUrl, enqueueStrategy);
268
+ if (!allowed) {
269
+ droppedUrlEntries++;
270
+ logger?.debug(`Skipping sitemap URL ${item.loc} (parent ${source.url}): ${reason}.`);
271
+ continue;
272
+ }
273
+ }
232
274
  yield {
233
275
  ...item,
234
276
  originSitemapUrl: source.type === 'url'
@@ -237,6 +279,9 @@ export async function* parseSitemap(initialSources, proxyUrl, options) {
237
279
  };
238
280
  }
239
281
  }
282
+ if (droppedUrlEntries > 0 && source.type === 'url') {
283
+ logger?.warning(`Skipped ${droppedUrlEntries} URL(s) from sitemap ${source.url} not matching enqueue strategy '${enqueueStrategy}' (or using a non-http(s) scheme). Enable debug logs to see each skipped URL.`);
284
+ }
240
285
  }
241
286
  }
242
287
  /**
@@ -262,7 +307,7 @@ export class Sitemap {
262
307
  * @param url The domain URL to fetch the sitemap for.
263
308
  * @param proxyUrl A proxy to be used for fetching the sitemap file.
264
309
  */
265
- static async tryCommonNames(url, proxyUrl) {
310
+ static async tryCommonNames(url, proxyUrl, parseSitemapOptions) {
266
311
  const sitemapUrls = [];
267
312
  const sitemapUrl = new URL(url);
268
313
  sitemapUrl.search = '';
@@ -270,7 +315,7 @@ export class Sitemap {
270
315
  sitemapUrls.push(sitemapUrl.toString());
271
316
  sitemapUrl.pathname = '/sitemap.txt';
272
317
  sitemapUrls.push(sitemapUrl.toString());
273
- return Sitemap.load(sitemapUrls, proxyUrl, { reportNetworkErrors: false });
318
+ return Sitemap.load(sitemapUrls, proxyUrl, { reportNetworkErrors: false, ...parseSitemapOptions });
274
319
  }
275
320
  /**
276
321
  * Fetch sitemap content from given URL or URLs and return URLs of referenced pages.
@@ -285,8 +330,8 @@ export class Sitemap {
285
330
  * @param content XML sitemap content
286
331
  * @param proxyUrl URL of a proxy to be used for fetching sitemap contents
287
332
  */
288
- static async fromXmlString(content, proxyUrl) {
289
- return await this.parse([{ type: 'raw', content }], proxyUrl);
333
+ static async fromXmlString(content, proxyUrl, parseSitemapOptions) {
334
+ return await this.parse([{ type: 'raw', content }], proxyUrl, parseSitemapOptions);
290
335
  }
291
336
  static async parse(sources, proxyUrl, parseSitemapOptions) {
292
337
  const urls = [];
@@ -295,10 +340,129 @@ export class Sitemap {
295
340
  urls.push(item.loc);
296
341
  }
297
342
  }
298
- catch {
343
+ catch (e) {
344
+ parseSitemapOptions?.logger?.warning(`Sitemap.load: Failed to load sitemap, returning empty result. (${e})`);
299
345
  return new Sitemap([]);
300
346
  }
301
347
  return new Sitemap(urls);
302
348
  }
303
349
  }
304
- //# sourceMappingURL=sitemap.js.map
350
+ /**
351
+ * Given a list of URLs, discover related sitemap files for these domains by checking the `robots.txt` file,
352
+ * the default `sitemap.xml` & `sitemap.txt` files and the URLs themselves.
353
+ * @param `urls` The list of URLs to discover sitemaps for.
354
+ * @param `options` Options for sitemap discovery
355
+ * @returns An async iterable with the discovered sitemap URLs.
356
+ */
357
+ export async function* discoverValidSitemaps(urls, options = {}) {
358
+ const { proxyUrl, timeoutMillis = 60_000, signal: externalSignal, requestTimeoutMillis = 20_000, httpClient = new FetchHttpClient(), logger, } = options;
359
+ const controller = new AbortController();
360
+ const timeoutHandle = setTimeout(() => controller.abort(), timeoutMillis);
361
+ const onExternalAbort = () => controller.abort();
362
+ if (externalSignal) {
363
+ if (externalSignal.aborted) {
364
+ controller.abort();
365
+ }
366
+ else {
367
+ externalSignal.addEventListener('abort', onExternalAbort, { once: true });
368
+ }
369
+ }
370
+ const signal = controller.signal;
371
+ const sitemapUrls = new Set();
372
+ const addSitemapUrl = (url) => {
373
+ const sizeBefore = sitemapUrls.size;
374
+ sitemapUrls.add(url);
375
+ if (sitemapUrls.size > sizeBefore) {
376
+ return url;
377
+ }
378
+ return undefined;
379
+ };
380
+ const urlExists = async (url) => {
381
+ if (!httpClient) {
382
+ return false;
383
+ }
384
+ try {
385
+ const response = await httpClient.sendRequest(new Request(url, { method: 'HEAD' }), {
386
+ proxyUrl,
387
+ timeoutMillis: requestTimeoutMillis,
388
+ signal,
389
+ });
390
+ return response.status >= 200 && response.status < 400;
391
+ }
392
+ catch {
393
+ return false;
394
+ }
395
+ };
396
+ const discoverSitemapsForDomainUrls = async function* (hostname, domainUrls) {
397
+ if (!hostname) {
398
+ return;
399
+ }
400
+ try {
401
+ const robotsFile = await RobotsTxtFile.find(domainUrls[0], {
402
+ proxyUrl,
403
+ timeoutMillis: requestTimeoutMillis,
404
+ signal,
405
+ httpClient,
406
+ logger,
407
+ });
408
+ // Surface all referenced sitemaps, including cross-host; scoping happens at load time.
409
+ for (const sitemapUrl of robotsFile.getSitemaps({ enqueueStrategy: 'all' })) {
410
+ if (addSitemapUrl(sitemapUrl)) {
411
+ yield sitemapUrl;
412
+ }
413
+ }
414
+ }
415
+ catch (err) {
416
+ logger?.warning(`Failed to fetch robots.txt file for ${hostname}`, { error: err });
417
+ }
418
+ const sitemapUrl = domainUrls.find((url) => /sitemap(?:_index)?\.(?:xml|txt)(?:\.gz)?$/i.test(url));
419
+ if (sitemapUrl !== undefined) {
420
+ if (addSitemapUrl(sitemapUrl)) {
421
+ yield sitemapUrl;
422
+ }
423
+ }
424
+ else {
425
+ const firstUrl = new URL(domainUrls[0]);
426
+ const possibleSitemapPathnames = ['/sitemap.xml', '/sitemap.txt', '/sitemap_index.xml'];
427
+ const candidateSitemapUrls = possibleSitemapPathnames.map((pathname) => {
428
+ firstUrl.pathname = pathname;
429
+ return firstUrl.toString();
430
+ });
431
+ const candidateResults = await Promise.allSettled(candidateSitemapUrls.map(urlExists));
432
+ for (const [index, result] of candidateResults.entries()) {
433
+ const candidateSitemapUrl = candidateSitemapUrls[index];
434
+ if (result.status === 'fulfilled') {
435
+ if (result.value && addSitemapUrl(candidateSitemapUrl)) {
436
+ yield candidateSitemapUrl;
437
+ }
438
+ }
439
+ else {
440
+ logger?.debug(`Failed to check sitemap candidate ${candidateSitemapUrl} for ${hostname}`, {
441
+ error: result.reason,
442
+ });
443
+ }
444
+ }
445
+ }
446
+ };
447
+ const groupedUrls = urls.reduce((acc, url) => {
448
+ const hostname = new URL(url)?.hostname ?? '';
449
+ acc[hostname] ??= [];
450
+ acc[hostname].push(url);
451
+ return acc;
452
+ }, {});
453
+ const iterables = Object.entries(groupedUrls).map(([hostname, domainUrls]) => discoverSitemapsForDomainUrls(hostname, domainUrls));
454
+ const discoveredUrls = new Set();
455
+ try {
456
+ for await (const url of mergeAsyncIterables(...iterables)) {
457
+ if (discoveredUrls.has(url)) {
458
+ continue;
459
+ }
460
+ discoveredUrls.add(url);
461
+ yield url;
462
+ }
463
+ }
464
+ finally {
465
+ clearTimeout(timeoutHandle);
466
+ externalSignal?.removeEventListener('abort', onExternalAbort);
467
+ }
468
+ }
@@ -488,5 +488,4 @@ export declare const DISCORD_REGEX_GLOBAL: RegExp;
488
488
  * so that the caller doesn't need to parse the HTML document again, if needed.
489
489
  * @return An object with the social handles.
490
490
  */
491
- export declare function parseHandlesFromHtml(html: string, data?: Record<string, unknown> | null): SocialHandles;
492
- //# sourceMappingURL=social.d.ts.map
491
+ export declare function parseHandlesFromHtml(html: string, data?: Record<string, unknown> | null): Promise<SocialHandles>;
@@ -1,7 +1,9 @@
1
- import * as cheerio from 'cheerio';
2
1
  import { htmlToText } from './cheerio.js';
3
2
  // Regex inspired by https://zapier.com/blog/extract-links-email-phone-regex/
4
- const EMAIL_REGEX_STRING = '(?:[a-z0-9!#$%&\'*+/=?^_`{|}~-]+(?:\\.[a-z0-9!#$%&\'*+/=?^_`{|}~-]+)*|"(?:[\x01-\x08\x0b\x0c\x0e-\x1f\x21\x23-\x5b\x5d-\x7f]|\\\\[\x01-\x09\x0b\x0c\x0e-\x7f])*")@(?:(?:[a-z0-9](?:[a-z0-9-]*[a-z0-9])?\\.)+[a-z0-9](?:[a-z0-9-]*[a-z0-9])?|\\[(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?)\\.){3}(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?|[a-z0-9-]*[a-z0-9]:(?:[\x01-\x08\x0b\x0c\x0e-\x1f\x21-\x5a\x53-\x7f]|\\\\[\x01-\x09\x0b\x0c\x0e-\x7f])+)\\])';
3
+ // The dot-atom local part and domain labels use RFC 5321 length bounds ({1,64}, {0,62})
4
+ // instead of unbounded quantifiers to avoid quadratic backtracking (ReDoS) on long
5
+ // dotted or hyphenated inputs, e.g. text scraped by parseHandlesFromHtml().
6
+ const EMAIL_REGEX_STRING = '(?:[a-z0-9!#$%&\'*+/=?^_`{|}~-]{1,64}(?:\\.[a-z0-9!#$%&\'*+/=?^_`{|}~-]{1,64}){0,32}|"(?:[\x01-\x08\x0b\x0c\x0e-\x1f\x21\x23-\x5b\x5d-\x7f]|\\\\[\x01-\x09\x0b\x0c\x0e-\x7f])*")@(?:(?:[a-z0-9](?:[a-z0-9-]{0,62}[a-z0-9])?\\.)+[a-z0-9](?:[a-z0-9-]{0,62}[a-z0-9])?|\\[(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?)\\.){3}(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?|[a-z0-9-]*[a-z0-9]:(?:[\x01-\x08\x0b\x0c\x0e-\x1f\x21-\x5a\x53-\x7f]|\\\\[\x01-\x09\x0b\x0c\x0e-\x7f])+)\\])';
5
7
  /**
6
8
  * Regular expression to exactly match a single email address.
7
9
  * It has the following form: `/^...$/i`.
@@ -587,7 +589,8 @@ export const DISCORD_REGEX_GLOBAL = new RegExp(DISCORD_REGEX_STRING, 'ig');
587
589
  * so that the caller doesn't need to parse the HTML document again, if needed.
588
590
  * @return An object with the social handles.
589
591
  */
590
- export function parseHandlesFromHtml(html, data = null) {
592
+ export async function parseHandlesFromHtml(html, data = null) {
593
+ const cheerio = await import('cheerio');
591
594
  const result = {
592
595
  emails: [],
593
596
  phones: [],
@@ -606,7 +609,7 @@ export function parseHandlesFromHtml(html, data = null) {
606
609
  const $ = cheerio.load(html, { xml: { decodeEntities: true } });
607
610
  if (data)
608
611
  data.$ = $;
609
- const text = htmlToText($);
612
+ const text = await htmlToText($);
610
613
  if (data)
611
614
  data.text = text;
612
615
  // NOTE: we need to parse each text separately, orherwise we might concatenate unrelated texts
@@ -644,4 +647,3 @@ export function parseHandlesFromHtml(html, data = null) {
644
647
  }
645
648
  return result;
646
649
  }
647
- //# sourceMappingURL=social.js.map
@@ -1,4 +1,73 @@
1
- export type SearchParams = string | URLSearchParams | Record<string, string | number | boolean | null | undefined>;
1
+ import type { SearchParams } from '@crawlee/types';
2
+ /**
3
+ * The different enqueueing strategies available.
4
+ *
5
+ * Depending on the strategy you select, we will only check certain parts of the URLs found. Here is a diagram of each URL part and their name:
6
+ *
7
+ * ```md
8
+ * Protocol Domain
9
+ * ┌────┐ ┌─────────┐
10
+ * https://example.crawlee.dev/...
11
+ * │ └─────────────────┤
12
+ * │ Hostname │
13
+ * │ │
14
+ * └─────────────────────────┘
15
+ * Origin
16
+ *```
17
+ *
18
+ * - The `Protocol` is usually `http` or `https`
19
+ * - The `Domain` represents the path without any possible subdomains to a website. For example, `crawlee.dev` is the domain of `https://example.crawlee.dev/`
20
+ * - The `Hostname` is the full path to a website, including any subdomains. For example, `example.crawlee.dev` is the hostname of `https://example.crawlee.dev/`
21
+ * - The `Origin` is the combination of the `Protocol` and `Hostname`. For example, `https://example.crawlee.dev` is the origin of `https://example.crawlee.dev/`
22
+ */
23
+ export declare enum EnqueueStrategy {
24
+ /**
25
+ * Matches any URLs found
26
+ */
27
+ All = "all",
28
+ /**
29
+ * Matches any URLs that have the same hostname.
30
+ * For example, `https://wow.example.com/hello` will be matched for a base url of `https://wow.example.com/`, but
31
+ * `https://example.com/hello` will not be matched.
32
+ *
33
+ * > This strategy will match both `http` and `https` protocols regardless of the base URL protocol.
34
+ */
35
+ SameHostname = "same-hostname",
36
+ /**
37
+ * Matches any URLs that have the same domain as the base URL.
38
+ * For example, `https://wow.an.example.com` and `https://example.com` will both be matched for a base url of
39
+ * `https://example.com`.
40
+ *
41
+ * > This strategy will match both `http` and `https` protocols regardless of the base URL protocol.
42
+ */
43
+ SameDomain = "same-domain",
44
+ /**
45
+ * Matches any URLs that have the same hostname and protocol.
46
+ * For example, `https://wow.example.com/hello` will be matched for a base url of `https://wow.example.com/`, but
47
+ * `http://wow.example.com/hello` will not be matched.
48
+ *
49
+ * > This strategy will ensure the protocol of the base URL is the same as the protocol of the URL to be enqueued.
50
+ */
51
+ SameOrigin = "same-origin"
52
+ }
53
+ /** Reusable suffix for log messages explaining why a non-`http(s)` URL was rejected. */
54
+ export declare const UNSUPPORTED_SCHEME_MESSAGE = "unsupported URL scheme (only http and https are allowed)";
55
+ /**
56
+ * Check whether `target` matches `origin` under the given enqueue `strategy`. The URL scheme is not
57
+ * considered here (use {@link filterUrl} for the combined scheme + strategy check).
58
+ *
59
+ * The `enqueueLinks` implementation in `@crawlee/core` matches the same strategies via glob patterns
60
+ * (see `packages/core/src/enqueue_links/enqueue_links.ts`) — keep the two in sync when changing either.
61
+ */
62
+ export declare function matchesEnqueueStrategy(strategy: EnqueueStrategy | `${EnqueueStrategy}`, target: URL, origin: URL): boolean;
63
+ /**
64
+ * Check whether `target` may be enqueued under `strategy` relative to `origin`: it must use an `http(s)`
65
+ * scheme and match the strategy. On rejection, `reason` is a human-readable message for log output.
66
+ */
67
+ export declare function filterUrl(target: string | URL, origin: string | URL, strategy: EnqueueStrategy | `${EnqueueStrategy}`): {
68
+ allowed: boolean;
69
+ reason?: string;
70
+ };
2
71
  /**
3
72
  * Appends search (query string) parameters to a URL, replacing the original value (if any).
4
73
  *
@@ -7,4 +76,3 @@ export type SearchParams = string | URLSearchParams | Record<string, string | nu
7
76
  * @internal
8
77
  */
9
78
  export declare function applySearchParams(url: URL, searchParams: SearchParams | undefined): void;
10
- //# sourceMappingURL=url.d.ts.map