@crawlee/core 4.0.0-beta.99 → 4.0.0-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. package/README.md +1 -1
  2. package/configuration.d.ts +16 -47
  3. package/configuration.js +13 -25
  4. package/debug.js +4 -4
  5. package/errors.d.ts +28 -38
  6. package/errors.js +33 -47
  7. package/events/event_manager.d.ts +2 -2
  8. package/events/event_manager.js +7 -6
  9. package/events/index.d.ts +1 -0
  10. package/events/local_event_manager.d.ts +1 -8
  11. package/events/local_event_manager.js +13 -13
  12. package/events/system_info.d.ts +38 -0
  13. package/index.d.ts +2 -8
  14. package/index.js +4 -8
  15. package/internal.d.ts +8 -0
  16. package/internal.js +9 -0
  17. package/log.d.ts +10 -11
  18. package/log.js +52 -20
  19. package/memory-storage/memory-storage.d.ts +15 -18
  20. package/memory-storage/memory-storage.js +80 -58
  21. package/memory-storage/resource-clients/dataset.d.ts +1 -6
  22. package/memory-storage/resource-clients/dataset.js +23 -31
  23. package/memory-storage/resource-clients/key-value-store.d.ts +1 -10
  24. package/memory-storage/resource-clients/key-value-store.js +43 -67
  25. package/memory-storage/resource-clients/request-queue.d.ts +1 -42
  26. package/memory-storage/resource-clients/request-queue.js +109 -117
  27. package/owned_or_injected.d.ts +1 -3
  28. package/owned_or_injected.js +17 -17
  29. package/package.json +17 -20
  30. package/proxy_configuration.d.ts +21 -26
  31. package/proxy_configuration.js +35 -25
  32. package/recoverable_state.d.ts +104 -47
  33. package/recoverable_state.js +199 -74
  34. package/request.d.ts +20 -107
  35. package/request.js +78 -244
  36. package/serialization.js +17 -16
  37. package/service_locator.d.ts +22 -10
  38. package/service_locator.js +59 -48
  39. package/storages/batched_adds.d.ts +37 -0
  40. package/storages/batched_adds.js +73 -0
  41. package/storages/dataset.d.ts +13 -8
  42. package/storages/dataset.js +149 -40
  43. package/storages/index.d.ts +4 -4
  44. package/storages/index.js +2 -4
  45. package/storages/key_value_store.d.ts +16 -35
  46. package/storages/key_value_store.js +223 -110
  47. package/storages/key_value_store_codec.js +6 -11
  48. package/storages/request_dedup_cache.d.ts +1 -4
  49. package/storages/request_dedup_cache.js +15 -15
  50. package/storages/request_list.d.ts +9 -104
  51. package/storages/request_list.js +236 -233
  52. package/storages/request_loader.d.ts +49 -18
  53. package/storages/request_loader.js +36 -1
  54. package/storages/request_manager.d.ts +86 -0
  55. package/storages/request_manager_tandem.d.ts +14 -38
  56. package/storages/request_manager_tandem.js +67 -64
  57. package/storages/request_queue.d.ts +23 -50
  58. package/storages/request_queue.js +371 -226
  59. package/storages/storage_instance_manager.d.ts +2 -4
  60. package/storages/storage_instance_manager.js +21 -21
  61. package/storages/storage_stats.d.ts +1 -1
  62. package/storages/storage_stats.js +4 -4
  63. package/storages/transaction.d.ts +270 -0
  64. package/storages/transaction.js +296 -0
  65. package/storages/utils.d.ts +6 -3
  66. package/storages/utils.js +11 -2
  67. package/system-info/runtime.js +7 -7
  68. package/url.d.ts +9 -0
  69. package/url.js +11 -0
  70. package/validators.d.ts +23 -25
  71. package/validators.js +14 -25
  72. package/autoscaling/autoscaled_pool.d.ts +0 -213
  73. package/autoscaling/autoscaled_pool.js +0 -378
  74. package/autoscaling/client_load_signal.d.ts +0 -59
  75. package/autoscaling/client_load_signal.js +0 -73
  76. package/autoscaling/concurrency_system.d.ts +0 -283
  77. package/autoscaling/concurrency_system.js +0 -350
  78. package/autoscaling/cpu_load_signal.d.ts +0 -44
  79. package/autoscaling/cpu_load_signal.js +0 -46
  80. package/autoscaling/event_loop_load_signal.d.ts +0 -54
  81. package/autoscaling/event_loop_load_signal.js +0 -60
  82. package/autoscaling/index.d.ts +0 -9
  83. package/autoscaling/index.js +0 -9
  84. package/autoscaling/load_signal.d.ts +0 -99
  85. package/autoscaling/load_signal.js +0 -103
  86. package/autoscaling/memory_load_signal.d.ts +0 -56
  87. package/autoscaling/memory_load_signal.js +0 -106
  88. package/autoscaling/snapshotter.d.ts +0 -87
  89. package/autoscaling/snapshotter.js +0 -67
  90. package/autoscaling/system_status.d.ts +0 -161
  91. package/autoscaling/system_status.js +0 -139
  92. package/autoscaling/weighted_avg.d.ts +0 -5
  93. package/autoscaling/weighted_avg.js +0 -14
  94. package/cookie_utils.d.ts +0 -44
  95. package/cookie_utils.js +0 -122
  96. package/crawlers/context_pipeline.d.ts +0 -70
  97. package/crawlers/context_pipeline.js +0 -122
  98. package/crawlers/crawler_commons.d.ts +0 -257
  99. package/crawlers/crawler_commons.js +0 -107
  100. package/crawlers/error_snapshotter.d.ts +0 -59
  101. package/crawlers/error_snapshotter.js +0 -117
  102. package/crawlers/error_tracker.d.ts +0 -54
  103. package/crawlers/error_tracker.js +0 -308
  104. package/crawlers/index.d.ts +0 -5
  105. package/crawlers/index.js +0 -5
  106. package/crawlers/internals/types.d.ts +0 -7
  107. package/crawlers/statistics.d.ts +0 -209
  108. package/crawlers/statistics.js +0 -350
  109. package/enqueue_links/enqueue_links.d.ts +0 -264
  110. package/enqueue_links/enqueue_links.js +0 -271
  111. package/enqueue_links/index.d.ts +0 -2
  112. package/enqueue_links/index.js +0 -2
  113. package/enqueue_links/shared.d.ts +0 -83
  114. package/enqueue_links/shared.js +0 -221
  115. package/router.d.ts +0 -309
  116. package/router.js +0 -309
  117. package/session_pool/consts.d.ts +0 -3
  118. package/session_pool/consts.js +0 -3
  119. package/session_pool/errors.d.ts +0 -7
  120. package/session_pool/errors.js +0 -11
  121. package/session_pool/fingerprint.d.ts +0 -9
  122. package/session_pool/fingerprint.js +0 -30
  123. package/session_pool/index.d.ts +0 -4
  124. package/session_pool/index.js +0 -4
  125. package/session_pool/session.d.ts +0 -161
  126. package/session_pool/session.js +0 -218
  127. package/session_pool/session_pool.d.ts +0 -246
  128. package/session_pool/session_pool.js +0 -386
  129. package/storages/access_checking.d.ts +0 -12
  130. package/storages/access_checking.js +0 -17
  131. package/storages/sitemap_request_loader.d.ts +0 -249
  132. package/storages/sitemap_request_loader.js +0 -432
  133. /package/{crawlers/internals/types.js → events/system_info.js} +0 -0
@@ -1,271 +0,0 @@
1
- import ow from 'ow';
2
- import { getDomain } from 'tldts';
3
- import { Request } from '../request.js';
4
- import { serviceLocator } from '../service_locator.js';
5
- import { applyRequestTransform, constructGlobObjectsFromGlobs, constructRegExpObjectsFromPseudoUrls, constructRegExpObjectsFromRegExps, createRequestOptions, filterRequestOptionsByPatterns, } from './shared.js';
6
- /**
7
- * The different enqueueing strategies available.
8
- *
9
- * Depending on the strategy you select, we will only check certain parts of the URLs found. Here is a diagram of each URL part and their name:
10
- *
11
- * ```md
12
- * Protocol Domain
13
- * ┌────┐ ┌─────────┐
14
- * https://example.crawlee.dev/...
15
- * │ └─────────────────┤
16
- * │ Hostname │
17
- * │ │
18
- * └─────────────────────────┘
19
- * Origin
20
- *```
21
- *
22
- * - The `Protocol` is usually `http` or `https`
23
- * - The `Domain` represents the path without any possible subdomains to a website. For example, `crawlee.dev` is the domain of `https://example.crawlee.dev/`
24
- * - The `Hostname` is the full path to a website, including any subdomains. For example, `example.crawlee.dev` is the hostname of `https://example.crawlee.dev/`
25
- * - The `Origin` is the combination of the `Protocol` and `Hostname`. For example, `https://example.crawlee.dev` is the origin of `https://example.crawlee.dev/`
26
- */
27
- export var EnqueueStrategy;
28
- (function (EnqueueStrategy) {
29
- /**
30
- * Matches any URLs found
31
- */
32
- EnqueueStrategy["All"] = "all";
33
- /**
34
- * Matches any URLs that have the same hostname.
35
- * For example, `https://wow.example.com/hello` will be matched for a base url of `https://wow.example.com/`, but
36
- * `https://example.com/hello` will not be matched.
37
- *
38
- * > This strategy will match both `http` and `https` protocols regardless of the base URL protocol.
39
- */
40
- EnqueueStrategy["SameHostname"] = "same-hostname";
41
- /**
42
- * Matches any URLs that have the same domain as the base URL.
43
- * For example, `https://wow.an.example.com` and `https://example.com` will both be matched for a base url of
44
- * `https://example.com`.
45
- *
46
- * > This strategy will match both `http` and `https` protocols regardless of the base URL protocol.
47
- */
48
- EnqueueStrategy["SameDomain"] = "same-domain";
49
- /**
50
- * Matches any URLs that have the same hostname and protocol.
51
- * For example, `https://wow.example.com/hello` will be matched for a base url of `https://wow.example.com/`, but
52
- * `http://wow.example.com/hello` will not be matched.
53
- *
54
- * > This strategy will ensure the protocol of the base URL is the same as the protocol of the URL to be enqueued.
55
- */
56
- EnqueueStrategy["SameOrigin"] = "same-origin";
57
- })(EnqueueStrategy || (EnqueueStrategy = {}));
58
- /**
59
- * This function enqueues the urls provided to the {@link RequestQueue} provided. If you want to automatically find and enqueue links,
60
- * you should use the context-aware `enqueueLinks` function provided on the crawler contexts.
61
- *
62
- * Optionally, the function allows you to filter the target links' URLs using an array of globs or regular expressions
63
- * and override settings of the enqueued {@link Request} objects.
64
- *
65
- * **Example usage**
66
- *
67
- * ```javascript
68
- * await enqueueLinks({
69
- * urls: aListOfFoundUrls,
70
- * requestManager,
71
- * selector: 'a.product-detail',
72
- * globs: [
73
- * 'https://www.example.com/handbags/*',
74
- * 'https://www.example.com/purses/*'
75
- * ],
76
- * });
77
- * ```
78
- *
79
- * @param options All `enqueueLinks()` parameters are passed via an options object.
80
- * @returns Promise that resolves to {@link BatchAddRequestsResult} object.
81
- */
82
- export async function enqueueLinks(options) {
83
- if (!options || Object.keys(options).length === 0) {
84
- throw new RangeError([
85
- 'enqueueLinks() was called without the required options. You can only do that when you use the `crawlingContext.enqueueLinks()` method in request handlers.',
86
- 'Check out our guide on how to use enqueueLinks() here: https://crawlee.dev/js/docs/examples/crawl-relative-links',
87
- ].join('\n'));
88
- }
89
- ow(options, ow.object.exactShape({
90
- urls: ow.array.ofType(ow.string),
91
- requestManager: ow.object.hasKeys('addRequestsBatched'),
92
- robotsTxtFile: ow.optional.object.hasKeys('isAllowed'),
93
- respectRobotsTxtFile: ow.optional.any(ow.boolean, ow.object.exactShape({ userAgent: ow.optional.string })),
94
- onSkippedRequest: ow.optional.function,
95
- forefront: ow.optional.boolean,
96
- skipNavigation: ow.optional.boolean,
97
- sessionId: ow.optional.string,
98
- limit: ow.optional.number,
99
- selector: ow.optional.string,
100
- baseUrl: ow.optional.string,
101
- userData: ow.optional.object,
102
- label: ow.optional.string,
103
- pseudoUrls: ow.optional.array.ofType(ow.any(ow.string, ow.object.hasKeys('purl'))),
104
- globs: ow.optional.array.ofType(ow.any(ow.string, ow.object.hasKeys('glob'))),
105
- exclude: ow.optional.array.ofType(ow.any(ow.string, ow.regExp, ow.object.hasKeys('glob'), ow.object.hasKeys('regexp'))),
106
- regexps: ow.optional.array.ofType(ow.any(ow.regExp, ow.object.hasKeys('regexp'))),
107
- transformRequestFunction: ow.optional.function,
108
- strategy: ow.optional.string.oneOf(Object.values(EnqueueStrategy)),
109
- waitForAllRequestsToBeAdded: ow.optional.boolean,
110
- }));
111
- const { requestManager, limit, urls,
112
- // oxlint-disable-next-line typescript/no-deprecated -- still accepted for backwards compat
113
- pseudoUrls, exclude, globs, regexps, transformRequestFunction, forefront, waitForAllRequestsToBeAdded, robotsTxtFile, onSkippedRequest, } = options;
114
- const urlExcludePatternObjects = [];
115
- const urlPatternObjects = [];
116
- if (exclude?.length) {
117
- for (const excl of exclude) {
118
- if (typeof excl === 'string' || 'glob' in excl) {
119
- urlExcludePatternObjects.push(...constructGlobObjectsFromGlobs([excl]));
120
- }
121
- else if (excl instanceof RegExp || 'regexp' in excl) {
122
- urlExcludePatternObjects.push(...constructRegExpObjectsFromRegExps([excl]));
123
- }
124
- }
125
- }
126
- if (pseudoUrls?.length) {
127
- serviceLocator.getLogger().deprecated('`pseudoUrls` option is deprecated, use `globs` or `regexps` instead');
128
- urlPatternObjects.push(...constructRegExpObjectsFromPseudoUrls(pseudoUrls));
129
- }
130
- if (globs?.length) {
131
- urlPatternObjects.push(...constructGlobObjectsFromGlobs(globs));
132
- }
133
- if (regexps?.length) {
134
- urlPatternObjects.push(...constructRegExpObjectsFromRegExps(regexps));
135
- }
136
- if (!urlPatternObjects.length) {
137
- options.strategy ??= EnqueueStrategy.SameHostname;
138
- }
139
- const enqueueStrategyPatterns = [];
140
- if (options.baseUrl) {
141
- const url = new URL(options.baseUrl);
142
- switch (options.strategy) {
143
- case EnqueueStrategy.SameHostname:
144
- // We need to get the origin of the passed in domain in the event someone sets baseUrl
145
- // to an url like https://example.com/deep/default/path and one of the found urls is an
146
- // absolute relative path (/path/to/page)
147
- enqueueStrategyPatterns.push({ glob: ignoreHttpSchema(`${url.origin}/**`) });
148
- break;
149
- case EnqueueStrategy.SameDomain: {
150
- // Get the actual hostname from the base url
151
- const baseUrlHostname = getDomain(url.hostname, { mixedInputs: false });
152
- if (baseUrlHostname) {
153
- // We have a hostname, so we can use it to match all links on the page that point to it and any subdomains of it
154
- url.hostname = baseUrlHostname;
155
- enqueueStrategyPatterns.push({ glob: ignoreHttpSchema(`${url.origin.replace(baseUrlHostname, `*.${baseUrlHostname}`)}/**`) }, { glob: ignoreHttpSchema(`${url.origin}/**`) });
156
- }
157
- else {
158
- // We don't have a hostname (can happen for ips for instance), so reproduce the same behavior
159
- // as SameDomainAndSubdomain
160
- enqueueStrategyPatterns.push({ glob: ignoreHttpSchema(`${url.origin}/**`) });
161
- }
162
- break;
163
- }
164
- case EnqueueStrategy.SameOrigin: {
165
- // The same behavior as SameHostname, but respecting the protocol of the URL
166
- enqueueStrategyPatterns.push({ glob: `${url.origin}/**` });
167
- break;
168
- }
169
- case EnqueueStrategy.All:
170
- default:
171
- enqueueStrategyPatterns.push({ glob: `http{s,}://**` });
172
- break;
173
- }
174
- }
175
- async function reportSkippedRequests(skippedRequests, reason) {
176
- if (onSkippedRequest && skippedRequests.length > 0) {
177
- await Promise.all(skippedRequests.map((request) => {
178
- return onSkippedRequest({
179
- url: request.url,
180
- reason: request.skippedReason ?? reason,
181
- });
182
- }));
183
- }
184
- }
185
- let requestOptions = createRequestOptions(urls, options);
186
- if (robotsTxtFile && options.respectRobotsTxtFile !== false) {
187
- const robotsUserAgent = typeof options.respectRobotsTxtFile === 'object' ? (options.respectRobotsTxtFile.userAgent ?? '*') : '*';
188
- const skippedRequests = [];
189
- requestOptions = requestOptions.filter((request) => {
190
- if (robotsTxtFile.isAllowed(request.url, robotsUserAgent)) {
191
- return true;
192
- }
193
- skippedRequests.push(request);
194
- return false;
195
- });
196
- await reportSkippedRequests(skippedRequests, 'robotsTxt');
197
- }
198
- async function createFilteredRequests() {
199
- const skippedRequests = [];
200
- // Step 1: Filter request options by exclude patterns, user patterns (globs/regexps), and strategy patterns.
201
- // Pattern-level options (label, userData, method, etc.) are merged during this step.
202
- let filteredOptions;
203
- if (urlPatternObjects.length === 0) {
204
- filteredOptions = filterRequestOptionsByPatterns(requestOptions, enqueueStrategyPatterns.length > 0 ? enqueueStrategyPatterns : undefined, urlExcludePatternObjects, options.strategy, (url) => skippedRequests.push(url));
205
- }
206
- else {
207
- // Filter by user patterns first (with exclude)
208
- const afterUserPatterns = filterRequestOptionsByPatterns(requestOptions, urlPatternObjects, urlExcludePatternObjects, options.strategy, (url) => skippedRequests.push(url));
209
- // ...then filter by the enqueue links strategy (making this an AND check)
210
- filteredOptions = filterRequestOptionsByPatterns(afterUserPatterns, enqueueStrategyPatterns.length > 0 ? enqueueStrategyPatterns : undefined, [], options.strategy, (url) => skippedRequests.push(url));
211
- }
212
- await reportSkippedRequests(skippedRequests.map((url) => ({ url })), 'filters');
213
- // Step 2: Apply transformRequestFunction on request options - it has the highest priority
214
- if (transformRequestFunction) {
215
- const skippedByTransform = [];
216
- filteredOptions = applyRequestTransform(filteredOptions, transformRequestFunction, (r) => skippedByTransform.push(r));
217
- await reportSkippedRequests(skippedByTransform, 'transform');
218
- }
219
- // Step 3: Create Request instances from the final request options
220
- return filteredOptions.map((opts) => new Request(opts));
221
- }
222
- const { addedRequests, requestsOverLimit } = await requestManager.addRequestsBatched(await createFilteredRequests(), {
223
- forefront,
224
- waitForAllRequestsToBeAdded,
225
- maxNewRequests: limit,
226
- });
227
- if (requestsOverLimit?.length !== undefined && requestsOverLimit.length > 0) {
228
- await reportSkippedRequests(requestsOverLimit.map((r) => ({ url: typeof r === 'string' ? r : r.url })), 'enqueueLimit');
229
- }
230
- return { processedRequests: addedRequests, unprocessedRequests: [] };
231
- }
232
- /**
233
- * @internal
234
- * This method helps resolve the baseUrl that will be used for filtering in {@link enqueueLinks}.
235
- * - If a user provides a base url, we always return it
236
- * - If a user specifies {@link EnqueueStrategy.All} strategy, they do not care if the newly found urls are on the original
237
- * request domain, or a redirected one
238
- * - In all other cases, we return the domain of the original request as that's the one we need to use for filtering
239
- */
240
- export function resolveBaseUrlForEnqueueLinksFiltering({ enqueueStrategy, finalRequestUrl, originalRequestUrl, userProvidedBaseUrl, }) {
241
- // User provided base url takes priority
242
- if (userProvidedBaseUrl) {
243
- return userProvidedBaseUrl;
244
- }
245
- const originalUrlOrigin = new URL(originalRequestUrl).origin;
246
- const finalUrlOrigin = new URL(finalRequestUrl ?? originalRequestUrl).origin;
247
- // We can assume users want to go off the domain in this case
248
- if (enqueueStrategy === EnqueueStrategy.All) {
249
- return finalUrlOrigin;
250
- }
251
- // If the user wants to ensure the same domain is accessed, regardless of subdomains, we check to ensure the domains match
252
- // Returning undefined here is intentional! If the domains don't match, having no baseUrl in enqueueLinks will cause it to not enqueue anything
253
- // which is the intended behavior (since we went off domain)
254
- if (enqueueStrategy === EnqueueStrategy.SameDomain) {
255
- const originalHostname = getDomain(originalUrlOrigin, { mixedInputs: false });
256
- const finalHostname = getDomain(finalUrlOrigin, { mixedInputs: false });
257
- if (originalHostname === finalHostname) {
258
- return finalUrlOrigin;
259
- }
260
- return undefined;
261
- }
262
- // Always enqueue urls that are from the same origin in all other cases, as the filtering happens on the original request url, even if there was a redirect
263
- // before actually finding the urls
264
- return originalUrlOrigin;
265
- }
266
- /**
267
- * Internal function that changes the enqueue globs to match both http and https
268
- */
269
- function ignoreHttpSchema(pattern) {
270
- return pattern.replace(/^(https?):\/\//, 'http{s,}://');
271
- }
@@ -1,2 +0,0 @@
1
- export * from './enqueue_links.js';
2
- export * from './shared.js';
@@ -1,2 +0,0 @@
1
- export * from './enqueue_links.js';
2
- export * from './shared.js';
@@ -1,83 +0,0 @@
1
- import type { Awaitable } from '@crawlee/types';
2
- import type { RequestOptions } from '../request.js';
3
- import type { EnqueueLinksOptions } from './enqueue_links.js';
4
- export { tryAbsoluteURL } from '@crawlee/utils';
5
- export type UrlPatternObject = {
6
- glob?: string;
7
- regexp?: RegExp;
8
- } & Pick<RequestOptions, 'method' | 'payload' | 'label' | 'userData' | 'headers'>;
9
- export type PseudoUrlObject = {
10
- purl: string;
11
- } & Pick<RequestOptions, 'method' | 'payload' | 'label' | 'userData' | 'headers'>;
12
- export type PseudoUrlInput = string | PseudoUrlObject;
13
- export type GlobObject = {
14
- glob: string;
15
- } & Pick<RequestOptions, 'method' | 'payload' | 'label' | 'userData' | 'headers'>;
16
- export type GlobInput = string | GlobObject;
17
- export type RegExpObject = {
18
- regexp: RegExp;
19
- } & Pick<RequestOptions, 'method' | 'payload' | 'label' | 'userData' | 'headers'>;
20
- export type RegExpInput = RegExp | RegExpObject;
21
- export type SkippedRequestReason = 'robotsTxt' | 'limit' | 'enqueueLimit' | 'filters' | 'transform' | 'redirect' | 'depth';
22
- export type SkippedRequestCallback = (args: {
23
- url: string;
24
- reason: SkippedRequestReason;
25
- }) => Awaitable<void>;
26
- /**
27
- * @ignore
28
- */
29
- export declare function updateEnqueueLinksPatternCache(item: GlobInput | RegExpInput | PseudoUrlInput, pattern: RegExpObject | GlobObject): void;
30
- /**
31
- * Helper factory used in the `enqueueLinks()` and enqueueLinksByClickingElements() function
32
- * to construct RegExps from PseudoUrl strings.
33
- * @ignore
34
- */
35
- export declare function constructRegExpObjectsFromPseudoUrls(pseudoUrls: readonly PseudoUrlInput[]): RegExpObject[];
36
- /**
37
- * Helper factory used in the `enqueueLinks()` and enqueueLinksByClickingElements() function
38
- * to construct Glob objects from Glob pattern strings.
39
- * @ignore
40
- */
41
- export declare function constructGlobObjectsFromGlobs(globs: readonly GlobInput[]): GlobObject[];
42
- /**
43
- * @internal
44
- */
45
- export declare function validateGlobPattern(glob: string): string;
46
- /**
47
- * Helper factory used in the `enqueueLinks()` and enqueueLinksByClickingElements() function
48
- * to check RegExps input and return valid RegExps.
49
- * @ignore
50
- */
51
- export declare function constructRegExpObjectsFromRegExps(regexps: readonly RegExpInput[]): RegExpObject[];
52
- /**
53
- * Filters request options by URL patterns and merges pattern-level options (label, userData, method, payload, headers)
54
- * from the first matching pattern into each RequestOptions entry.
55
- *
56
- * When `includePatterns` is empty/undefined, all options pass through (only exclude filtering applies).
57
- * @ignore
58
- */
59
- export declare function filterRequestOptionsByPatterns(requestOptions: RequestOptions[], includePatterns: UrlPatternObject[] | undefined, excludePatterns?: UrlPatternObject[], strategy?: EnqueueLinksOptions['strategy'], onSkippedUrl?: (url: string) => void): RequestOptions[];
60
- /**
61
- * @ignore
62
- */
63
- export declare function createRequestOptions(sources: readonly (string | Record<string, unknown>)[], options?: Pick<EnqueueLinksOptions, 'label' | 'userData' | 'baseUrl' | 'skipNavigation' | 'sessionId' | 'strategy'>): RequestOptions[];
64
- /**
65
- * Takes a {@link RequestOptions} object and changes its attributes in a desired way. This user-function is used
66
- * by {@link enqueueLinks} to modify request options before they are converted to {@link Request} instances.
67
- */
68
- export interface RequestTransform {
69
- /**
70
- * @param original Request options to be modified.
71
- * @returns The modified request options to enqueue, `'unchanged'` to keep the original options as-is,
72
- * or a falsy value / `'skip'` to exclude the request from the queue.
73
- */
74
- (original: RequestOptions): RequestOptions | false | undefined | null | 'skip' | 'unchanged';
75
- }
76
- /**
77
- * Applies a {@link RequestTransform} function to a list of request options.
78
- * Options for which the transform returns a falsy value are removed from the list.
79
- * @param onSkipped Called with the original request options when the transform returns a falsy value (i.e. the request is skipped).
80
- * @ignore
81
- * @internal
82
- */
83
- export declare function applyRequestTransform(requestOptions: RequestOptions[], transformFn: RequestTransform, onSkipped?: (requestOptions: RequestOptions) => void): RequestOptions[];
@@ -1,221 +0,0 @@
1
- import { URL } from 'node:url';
2
- import { Minimatch } from 'minimatch';
3
- import { purlToRegExp } from '@apify/pseudo_url';
4
- export { tryAbsoluteURL } from '@crawlee/utils';
5
- const MAX_ENQUEUE_LINKS_CACHE_SIZE = 1000;
6
- /**
7
- * To enable direct use of the Actor UI `globs`/`regexps`/`pseudoUrls` output while keeping high performance,
8
- * all the regexps from the output are only constructed once and kept in a cache
9
- * by the `enqueueLinks()` function.
10
- * @ignore
11
- */
12
- const enqueueLinksPatternCache = new Map();
13
- /**
14
- * @ignore
15
- */
16
- export function updateEnqueueLinksPatternCache(item, pattern) {
17
- enqueueLinksPatternCache.set(item, pattern);
18
- if (enqueueLinksPatternCache.size > MAX_ENQUEUE_LINKS_CACHE_SIZE) {
19
- const key = enqueueLinksPatternCache.keys().next().value;
20
- enqueueLinksPatternCache.delete(key);
21
- }
22
- }
23
- /**
24
- * Helper factory used in the `enqueueLinks()` and enqueueLinksByClickingElements() function
25
- * to construct RegExps from PseudoUrl strings.
26
- * @ignore
27
- */
28
- export function constructRegExpObjectsFromPseudoUrls(pseudoUrls) {
29
- return pseudoUrls.map((item) => {
30
- // Get pseudoUrl object from cache.
31
- let regexpObject = enqueueLinksPatternCache.get(item);
32
- if (regexpObject)
33
- return regexpObject;
34
- if (typeof item === 'string') {
35
- regexpObject = { regexp: purlToRegExp(item) };
36
- }
37
- else {
38
- const { purl, ...requestOptions } = item;
39
- regexpObject = { regexp: purlToRegExp(purl), ...requestOptions };
40
- }
41
- updateEnqueueLinksPatternCache(item, regexpObject);
42
- return regexpObject;
43
- });
44
- }
45
- /**
46
- * Helper factory used in the `enqueueLinks()` and enqueueLinksByClickingElements() function
47
- * to construct Glob objects from Glob pattern strings.
48
- * @ignore
49
- */
50
- export function constructGlobObjectsFromGlobs(globs) {
51
- return globs
52
- .filter((glob) => {
53
- // Skip possibly nullish, empty strings
54
- if (!glob) {
55
- return false;
56
- }
57
- if (typeof glob === 'string') {
58
- return glob.trim().length > 0;
59
- }
60
- if (glob.glob) {
61
- return glob.glob.trim().length > 0;
62
- }
63
- return false;
64
- })
65
- .map((item) => {
66
- // Get glob object from cache.
67
- let globObject = enqueueLinksPatternCache.get(item);
68
- if (globObject)
69
- return globObject;
70
- if (typeof item === 'string') {
71
- globObject = { glob: validateGlobPattern(item) };
72
- }
73
- else {
74
- const { glob, ...requestOptions } = item;
75
- globObject = { glob: validateGlobPattern(glob), ...requestOptions };
76
- }
77
- updateEnqueueLinksPatternCache(item, globObject);
78
- return globObject;
79
- });
80
- }
81
- /**
82
- * @internal
83
- */
84
- export function validateGlobPattern(glob) {
85
- const globTrimmed = glob.trim();
86
- if (globTrimmed.length === 0)
87
- throw new Error(`Cannot parse Glob pattern '${globTrimmed}': it must be an non-empty string`);
88
- return globTrimmed;
89
- }
90
- /**
91
- * Helper factory used in the `enqueueLinks()` and enqueueLinksByClickingElements() function
92
- * to check RegExps input and return valid RegExps.
93
- * @ignore
94
- */
95
- export function constructRegExpObjectsFromRegExps(regexps) {
96
- return regexps.map((item) => {
97
- // Get regexp object from cache.
98
- let regexpObject = enqueueLinksPatternCache.get(item);
99
- if (regexpObject)
100
- return regexpObject;
101
- if (item instanceof RegExp) {
102
- regexpObject = { regexp: item };
103
- }
104
- else {
105
- regexpObject = item;
106
- }
107
- updateEnqueueLinksPatternCache(item, regexpObject);
108
- return regexpObject;
109
- });
110
- }
111
- /**
112
- * Filters request options by URL patterns and merges pattern-level options (label, userData, method, payload, headers)
113
- * from the first matching pattern into each RequestOptions entry.
114
- *
115
- * When `includePatterns` is empty/undefined, all options pass through (only exclude filtering applies).
116
- * @ignore
117
- */
118
- export function filterRequestOptionsByPatterns(requestOptions, includePatterns, excludePatterns = [], strategy, onSkippedUrl) {
119
- const excludeMatchers = excludePatterns.map(createPatternObjectMatcher);
120
- const includeMatchers = includePatterns?.length ? includePatterns.map(createPatternObjectMatcher) : undefined;
121
- return requestOptions
122
- .filter(({ url }) => {
123
- const matchesExclude = excludeMatchers.some(({ match }) => match(url));
124
- if (matchesExclude) {
125
- onSkippedUrl?.(url);
126
- }
127
- return !matchesExclude;
128
- })
129
- .map((opts) => {
130
- if (!includeMatchers) {
131
- return { ...opts, enqueueStrategy: strategy };
132
- }
133
- for (const { match, glob, regexp, ...patternOptions } of includeMatchers) {
134
- if (match(opts.url)) {
135
- return { ...opts, ...patternOptions, enqueueStrategy: strategy };
136
- }
137
- }
138
- // didn't match any positive pattern
139
- onSkippedUrl?.(opts.url);
140
- return null;
141
- })
142
- .filter((opts) => opts !== null);
143
- }
144
- /**
145
- * @ignore
146
- */
147
- export function createRequestOptions(sources, options = {}) {
148
- return sources
149
- .map((src) => typeof src === 'string'
150
- ? { url: src, enqueueStrategy: options.strategy }
151
- : { ...src, enqueueStrategy: options.strategy })
152
- .filter(({ url }) => {
153
- try {
154
- return new URL(url, options.baseUrl).href;
155
- }
156
- catch (err) {
157
- return false;
158
- }
159
- })
160
- .map((requestOptions) => {
161
- requestOptions.url = new URL(requestOptions.url, options.baseUrl).href;
162
- requestOptions.userData ??= options.userData ?? {};
163
- if (typeof options.label === 'string') {
164
- requestOptions.userData = {
165
- ...requestOptions.userData,
166
- label: options.label,
167
- };
168
- }
169
- if (options.skipNavigation) {
170
- requestOptions.skipNavigation = true;
171
- }
172
- if (options.sessionId) {
173
- requestOptions.sessionId = options.sessionId;
174
- }
175
- return requestOptions;
176
- });
177
- }
178
- /**
179
- * @ignore
180
- */
181
- function createPatternObjectMatcher(urlPatternObject) {
182
- const { regexp, glob } = urlPatternObject;
183
- let match;
184
- if (regexp) {
185
- match = (url) => regexp.test(url);
186
- }
187
- else if (glob) {
188
- const m = new Minimatch(glob, { nocase: true });
189
- match = (url) => m.match(url);
190
- }
191
- else {
192
- match = () => false;
193
- }
194
- return { ...urlPatternObject, match };
195
- }
196
- /**
197
- * Applies a {@link RequestTransform} function to a list of request options.
198
- * Options for which the transform returns a falsy value are removed from the list.
199
- * @param onSkipped Called with the original request options when the transform returns a falsy value (i.e. the request is skipped).
200
- * @ignore
201
- * @internal
202
- */
203
- export function applyRequestTransform(requestOptions, transformFn, onSkipped) {
204
- return requestOptions
205
- .map((opts) => {
206
- const transformed = transformFn(opts);
207
- if (transformed === 'skip') {
208
- onSkipped?.(opts);
209
- return null;
210
- }
211
- if (transformed === 'unchanged') {
212
- return opts;
213
- }
214
- if (!transformed) {
215
- onSkipped?.(opts);
216
- return null;
217
- }
218
- return transformed;
219
- })
220
- .filter((r) => r !== null);
221
- }