@crawlee/core 4.0.0-beta.99 → 4.0.0-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. package/README.md +1 -1
  2. package/configuration.d.ts +16 -47
  3. package/configuration.js +13 -25
  4. package/debug.js +4 -4
  5. package/errors.d.ts +28 -38
  6. package/errors.js +33 -47
  7. package/events/event_manager.d.ts +2 -2
  8. package/events/event_manager.js +7 -6
  9. package/events/index.d.ts +1 -0
  10. package/events/local_event_manager.d.ts +1 -8
  11. package/events/local_event_manager.js +13 -13
  12. package/events/system_info.d.ts +38 -0
  13. package/index.d.ts +2 -8
  14. package/index.js +4 -8
  15. package/internal.d.ts +8 -0
  16. package/internal.js +9 -0
  17. package/log.d.ts +10 -11
  18. package/log.js +52 -20
  19. package/memory-storage/memory-storage.d.ts +15 -18
  20. package/memory-storage/memory-storage.js +80 -58
  21. package/memory-storage/resource-clients/dataset.d.ts +1 -6
  22. package/memory-storage/resource-clients/dataset.js +23 -31
  23. package/memory-storage/resource-clients/key-value-store.d.ts +1 -10
  24. package/memory-storage/resource-clients/key-value-store.js +43 -67
  25. package/memory-storage/resource-clients/request-queue.d.ts +1 -42
  26. package/memory-storage/resource-clients/request-queue.js +109 -117
  27. package/owned_or_injected.d.ts +1 -3
  28. package/owned_or_injected.js +17 -17
  29. package/package.json +17 -20
  30. package/proxy_configuration.d.ts +21 -26
  31. package/proxy_configuration.js +35 -25
  32. package/recoverable_state.d.ts +104 -47
  33. package/recoverable_state.js +199 -74
  34. package/request.d.ts +20 -107
  35. package/request.js +78 -244
  36. package/serialization.js +17 -16
  37. package/service_locator.d.ts +22 -10
  38. package/service_locator.js +59 -48
  39. package/storages/batched_adds.d.ts +37 -0
  40. package/storages/batched_adds.js +73 -0
  41. package/storages/dataset.d.ts +13 -8
  42. package/storages/dataset.js +149 -40
  43. package/storages/index.d.ts +4 -4
  44. package/storages/index.js +2 -4
  45. package/storages/key_value_store.d.ts +16 -35
  46. package/storages/key_value_store.js +223 -110
  47. package/storages/key_value_store_codec.js +6 -11
  48. package/storages/request_dedup_cache.d.ts +1 -4
  49. package/storages/request_dedup_cache.js +15 -15
  50. package/storages/request_list.d.ts +9 -104
  51. package/storages/request_list.js +236 -233
  52. package/storages/request_loader.d.ts +49 -18
  53. package/storages/request_loader.js +36 -1
  54. package/storages/request_manager.d.ts +86 -0
  55. package/storages/request_manager_tandem.d.ts +14 -38
  56. package/storages/request_manager_tandem.js +67 -64
  57. package/storages/request_queue.d.ts +23 -50
  58. package/storages/request_queue.js +371 -226
  59. package/storages/storage_instance_manager.d.ts +2 -4
  60. package/storages/storage_instance_manager.js +21 -21
  61. package/storages/storage_stats.d.ts +1 -1
  62. package/storages/storage_stats.js +4 -4
  63. package/storages/transaction.d.ts +270 -0
  64. package/storages/transaction.js +296 -0
  65. package/storages/utils.d.ts +6 -3
  66. package/storages/utils.js +11 -2
  67. package/system-info/runtime.js +7 -7
  68. package/url.d.ts +9 -0
  69. package/url.js +11 -0
  70. package/validators.d.ts +23 -25
  71. package/validators.js +14 -25
  72. package/autoscaling/autoscaled_pool.d.ts +0 -213
  73. package/autoscaling/autoscaled_pool.js +0 -378
  74. package/autoscaling/client_load_signal.d.ts +0 -59
  75. package/autoscaling/client_load_signal.js +0 -73
  76. package/autoscaling/concurrency_system.d.ts +0 -283
  77. package/autoscaling/concurrency_system.js +0 -350
  78. package/autoscaling/cpu_load_signal.d.ts +0 -44
  79. package/autoscaling/cpu_load_signal.js +0 -46
  80. package/autoscaling/event_loop_load_signal.d.ts +0 -54
  81. package/autoscaling/event_loop_load_signal.js +0 -60
  82. package/autoscaling/index.d.ts +0 -9
  83. package/autoscaling/index.js +0 -9
  84. package/autoscaling/load_signal.d.ts +0 -99
  85. package/autoscaling/load_signal.js +0 -103
  86. package/autoscaling/memory_load_signal.d.ts +0 -56
  87. package/autoscaling/memory_load_signal.js +0 -106
  88. package/autoscaling/snapshotter.d.ts +0 -87
  89. package/autoscaling/snapshotter.js +0 -67
  90. package/autoscaling/system_status.d.ts +0 -161
  91. package/autoscaling/system_status.js +0 -139
  92. package/autoscaling/weighted_avg.d.ts +0 -5
  93. package/autoscaling/weighted_avg.js +0 -14
  94. package/cookie_utils.d.ts +0 -44
  95. package/cookie_utils.js +0 -122
  96. package/crawlers/context_pipeline.d.ts +0 -70
  97. package/crawlers/context_pipeline.js +0 -122
  98. package/crawlers/crawler_commons.d.ts +0 -257
  99. package/crawlers/crawler_commons.js +0 -107
  100. package/crawlers/error_snapshotter.d.ts +0 -59
  101. package/crawlers/error_snapshotter.js +0 -117
  102. package/crawlers/error_tracker.d.ts +0 -54
  103. package/crawlers/error_tracker.js +0 -308
  104. package/crawlers/index.d.ts +0 -5
  105. package/crawlers/index.js +0 -5
  106. package/crawlers/internals/types.d.ts +0 -7
  107. package/crawlers/statistics.d.ts +0 -209
  108. package/crawlers/statistics.js +0 -350
  109. package/enqueue_links/enqueue_links.d.ts +0 -264
  110. package/enqueue_links/enqueue_links.js +0 -271
  111. package/enqueue_links/index.d.ts +0 -2
  112. package/enqueue_links/index.js +0 -2
  113. package/enqueue_links/shared.d.ts +0 -83
  114. package/enqueue_links/shared.js +0 -221
  115. package/router.d.ts +0 -309
  116. package/router.js +0 -309
  117. package/session_pool/consts.d.ts +0 -3
  118. package/session_pool/consts.js +0 -3
  119. package/session_pool/errors.d.ts +0 -7
  120. package/session_pool/errors.js +0 -11
  121. package/session_pool/fingerprint.d.ts +0 -9
  122. package/session_pool/fingerprint.js +0 -30
  123. package/session_pool/index.d.ts +0 -4
  124. package/session_pool/index.js +0 -4
  125. package/session_pool/session.d.ts +0 -161
  126. package/session_pool/session.js +0 -218
  127. package/session_pool/session_pool.d.ts +0 -246
  128. package/session_pool/session_pool.js +0 -386
  129. package/storages/access_checking.d.ts +0 -12
  130. package/storages/access_checking.js +0 -17
  131. package/storages/sitemap_request_loader.d.ts +0 -249
  132. package/storages/sitemap_request_loader.js +0 -432
  133. /package/{crawlers/internals/types.js → events/system_info.js} +0 -0
@@ -1,350 +0,0 @@
1
- import ow from 'ow';
2
- import { serviceLocator } from '../service_locator.js';
3
- import { KeyValueStore } from '../storages/key_value_store.js';
4
- import { ErrorTracker } from './error_tracker.js';
5
- /**
6
- * @ignore
7
- */
8
- class Job {
9
- lastRunAt = null;
10
- durationMillis;
11
- run() {
12
- this.lastRunAt = Date.now();
13
- }
14
- finish() {
15
- this.durationMillis = Date.now() - this.lastRunAt;
16
- return this.durationMillis;
17
- }
18
- }
19
- const errorTrackerConfig = {
20
- showErrorCode: true,
21
- showErrorName: true,
22
- showStackTrace: true,
23
- showFullStack: false,
24
- showErrorMessage: true,
25
- showFullMessage: false,
26
- };
27
- /**
28
- * The statistics class provides an interface to collecting and logging run
29
- * statistics for requests.
30
- *
31
- * All statistic information is saved on key value store
32
- * under the key `CRAWLEE_CRAWLER_STATISTICS_*`, persists between
33
- * migrations and abort/resurrect
34
- *
35
- * @category Crawlers
36
- */
37
- export class Statistics {
38
- static id = 0;
39
- /**
40
- * An error tracker for final retry errors.
41
- */
42
- errorTracker;
43
- /**
44
- * An error tracker for retry errors prior to the final retry.
45
- */
46
- errorTrackerRetry;
47
- /**
48
- * Statistic instance id.
49
- */
50
- id;
51
- /**
52
- * Current statistic state used for doing calculations on {@link Statistics.calculate} calls
53
- */
54
- state;
55
- /**
56
- * Contains the current retries histogram. Index 0 means 0 retries, index 2, 2 retries, and so on
57
- */
58
- requestRetryHistogram = [];
59
- keyValueStore = undefined;
60
- persistStateKey;
61
- logIntervalMillis;
62
- logMessage;
63
- listener;
64
- requestsInProgress = new Map();
65
- log;
66
- instanceStart;
67
- logInterval;
68
- _events;
69
- persistenceOptions;
70
- get events() {
71
- if (!this._events) {
72
- this._events = serviceLocator.getEventManager();
73
- }
74
- return this._events;
75
- }
76
- /**
77
- * @internal
78
- */
79
- constructor(options = {}) {
80
- ow(options, ow.object.exactShape({
81
- logIntervalSecs: ow.optional.number,
82
- logMessage: ow.optional.string,
83
- log: ow.optional.object,
84
- keyValueStore: ow.optional.object,
85
- persistenceOptions: ow.optional.object,
86
- saveErrorSnapshots: ow.optional.boolean,
87
- id: ow.optional.any(ow.number, ow.string),
88
- }));
89
- const { logIntervalSecs = 60, logMessage = 'Statistics', keyValueStore, persistenceOptions = {
90
- enable: true,
91
- }, saveErrorSnapshots = false, id, } = options;
92
- this.id = id ?? String(Statistics.id++);
93
- this.persistStateKey = `CRAWLEE_CRAWLER_STATISTICS_${this.id}`;
94
- this.log = (options.log ?? serviceLocator.getLogger()).child({ prefix: 'Statistics' });
95
- this.errorTracker = new ErrorTracker({ ...errorTrackerConfig, saveErrorSnapshots });
96
- this.errorTrackerRetry = new ErrorTracker({ ...errorTrackerConfig, saveErrorSnapshots });
97
- this.logIntervalMillis = logIntervalSecs * 1000;
98
- this.logMessage = logMessage;
99
- this.keyValueStore = keyValueStore;
100
- this.listener = this.persistState.bind(this);
101
- this.persistenceOptions = persistenceOptions;
102
- // initialize by "resetting"
103
- this.reset();
104
- }
105
- /**
106
- * Set the current statistic instance to pristine values
107
- */
108
- reset() {
109
- this.errorTracker.reset();
110
- this.errorTrackerRetry.reset();
111
- this.state = {
112
- requestsFinished: 0,
113
- requestsFailed: 0,
114
- requestsRetries: 0,
115
- requestsFailedPerMinute: 0,
116
- requestsFinishedPerMinute: 0,
117
- requestMinDurationMillis: Infinity,
118
- requestMaxDurationMillis: 0,
119
- requestTotalFailedDurationMillis: 0,
120
- requestTotalFinishedDurationMillis: 0,
121
- crawlerStartedAt: null,
122
- crawlerFinishedAt: null,
123
- statsPersistedAt: null,
124
- crawlerRuntimeMillis: 0,
125
- requestsWithStatusCode: {},
126
- errors: this.errorTracker.result,
127
- retryErrors: this.errorTrackerRetry.result,
128
- };
129
- this.requestRetryHistogram.length = 0;
130
- this.requestsInProgress.clear();
131
- this.instanceStart = Date.now();
132
- this.teardown();
133
- }
134
- /**
135
- * @param options - Override the persistence options provided in the constructor
136
- */
137
- async resetStore(options) {
138
- if (!this.persistenceOptions.enable && !options?.enable) {
139
- return;
140
- }
141
- if (!this.keyValueStore) {
142
- return;
143
- }
144
- await this.keyValueStore.setValue(this.persistStateKey, null);
145
- }
146
- /**
147
- * Increments the status code counter.
148
- */
149
- registerStatusCode(code) {
150
- const s = String(code);
151
- if (this.state.requestsWithStatusCode[s] === undefined) {
152
- this.state.requestsWithStatusCode[s] = 0;
153
- }
154
- this.state.requestsWithStatusCode[s]++;
155
- }
156
- /**
157
- * Starts a job
158
- * @ignore
159
- */
160
- startJob(id) {
161
- let job = this.requestsInProgress.get(id);
162
- if (!job)
163
- job = new Job();
164
- job.run();
165
- this.requestsInProgress.set(id, job);
166
- }
167
- /**
168
- * Mark job as finished and sets the state
169
- * @ignore
170
- */
171
- finishJob(id, retryCount) {
172
- const job = this.requestsInProgress.get(id);
173
- if (!job)
174
- return;
175
- const jobDurationMillis = job.finish();
176
- this.state.requestsFinished++;
177
- this.state.requestTotalFinishedDurationMillis += jobDurationMillis;
178
- this.saveRetryCountForJob(retryCount);
179
- if (jobDurationMillis < this.state.requestMinDurationMillis)
180
- this.state.requestMinDurationMillis = jobDurationMillis;
181
- if (jobDurationMillis > this.state.requestMaxDurationMillis)
182
- this.state.requestMaxDurationMillis = jobDurationMillis;
183
- this.requestsInProgress.delete(id);
184
- }
185
- /**
186
- * Mark job as failed and sets the state
187
- * @ignore
188
- */
189
- failJob(id, retryCount) {
190
- const job = this.requestsInProgress.get(id);
191
- if (!job)
192
- return;
193
- this.state.requestTotalFailedDurationMillis += job.finish();
194
- this.state.requestsFailed++;
195
- this.saveRetryCountForJob(retryCount);
196
- this.requestsInProgress.delete(id);
197
- }
198
- /**
199
- * Discards a started job without affecting the finished/failed counters, e.g. when a request
200
- * turns out to be skipped (robots.txt, enqueue strategy) after `startJob` was already called for it.
201
- * @ignore
202
- */
203
- discardJob(id) {
204
- this.requestsInProgress.delete(id);
205
- }
206
- /**
207
- * Calculate the current statistics
208
- */
209
- calculate() {
210
- const { requestsFailed, requestsFinished, requestTotalFailedDurationMillis, requestTotalFinishedDurationMillis, } = this.state;
211
- const totalMillis = Date.now() - this.instanceStart;
212
- const totalMinutes = totalMillis / 1000 / 60;
213
- return {
214
- requestAvgFailedDurationMillis: Math.round(requestTotalFailedDurationMillis / requestsFailed) || Infinity,
215
- requestAvgFinishedDurationMillis: Math.round(requestTotalFinishedDurationMillis / requestsFinished) || Infinity,
216
- requestsFinishedPerMinute: Math.round(requestsFinished / totalMinutes) || 0,
217
- requestsFailedPerMinute: Math.floor(requestsFailed / totalMinutes) || 0,
218
- requestTotalDurationMillis: requestTotalFinishedDurationMillis + requestTotalFailedDurationMillis,
219
- requestsTotal: requestsFailed + requestsFinished,
220
- crawlerRuntimeMillis: totalMillis,
221
- };
222
- }
223
- /**
224
- * Initializes the key value store for persisting the statistics,
225
- * displaying the current state in predefined intervals
226
- */
227
- async startCapturing() {
228
- this.keyValueStore ??= await KeyValueStore.open(null, { configuration: serviceLocator.getConfiguration() });
229
- if (this.state.crawlerStartedAt === null) {
230
- this.state.crawlerStartedAt = new Date();
231
- }
232
- if (this.persistenceOptions.enable) {
233
- await this.maybeLoadStatistics();
234
- this.events.on("persistState" /* EventType.PERSIST_STATE */, this.listener);
235
- }
236
- this.logInterval = setInterval(() => {
237
- this.log.info(this.logMessage, {
238
- ...this.calculate(),
239
- retryHistogram: this.requestRetryHistogram,
240
- });
241
- }, this.logIntervalMillis);
242
- }
243
- /**
244
- * Stops logging and remove event listeners, then persist
245
- */
246
- async stopCapturing() {
247
- this.teardown();
248
- this.state.crawlerFinishedAt = new Date();
249
- await this.persistState();
250
- }
251
- saveRetryCountForJob(retryCount) {
252
- if (retryCount > 0)
253
- this.state.requestsRetries++;
254
- this.requestRetryHistogram[retryCount] ??= 0;
255
- this.requestRetryHistogram[retryCount]++;
256
- }
257
- /**
258
- * Persist internal state to the key value store
259
- * @param options - Override the persistence options provided in the constructor
260
- */
261
- async persistState(options) {
262
- if (!this.persistenceOptions.enable && !options?.enable) {
263
- return;
264
- }
265
- // this might be called before startCapturing was called without using await, should not crash
266
- if (!this.keyValueStore) {
267
- return;
268
- }
269
- this.log.debug('Persisting state', { persistStateKey: this.persistStateKey });
270
- await this.keyValueStore
271
- .setValue(this.persistStateKey, this.toJSON())
272
- .catch((error) => this.log.warning(`Failed to persist the statistics to ${this.persistStateKey}`, { error }));
273
- }
274
- /**
275
- * Loads the current statistic from the key value store if any
276
- */
277
- async maybeLoadStatistics() {
278
- // this might be called before startCapturing was called without using await, should not crash
279
- if (!this.keyValueStore) {
280
- return;
281
- }
282
- const savedState = await this.keyValueStore.getValue(this.persistStateKey);
283
- if (!savedState)
284
- return;
285
- // We saw a run where the requestRetryHistogram was not iterable and crashed
286
- // the crawler. Adding some logging to monitor this problem in the future.
287
- if (!Array.isArray(savedState.requestRetryHistogram)) {
288
- this.log.warning('Received invalid state from Key-value store.', {
289
- persistStateKey: this.persistStateKey,
290
- state: savedState,
291
- });
292
- }
293
- this.log.debug('Recreating state from KeyValueStore', { persistStateKey: this.persistStateKey });
294
- // the `requestRetryHistogram` array might be very large, we could end up with
295
- // `RangeError: Maximum call stack size exceeded` if we use `a.push(...b)`
296
- savedState.requestRetryHistogram.forEach((idx) => this.requestRetryHistogram.push(idx));
297
- this.state.requestsFinished = savedState.requestsFinished;
298
- this.state.requestsFailed = savedState.requestsFailed;
299
- this.state.requestsRetries = savedState.requestsRetries;
300
- this.state.requestTotalFailedDurationMillis = savedState.requestTotalFailedDurationMillis;
301
- this.state.requestTotalFinishedDurationMillis = savedState.requestTotalFinishedDurationMillis;
302
- this.state.requestMinDurationMillis = savedState.requestMinDurationMillis;
303
- this.state.requestMaxDurationMillis = savedState.requestMaxDurationMillis;
304
- // persisted state uses ISO date strings
305
- this.state.crawlerFinishedAt = savedState.crawlerFinishedAt ? new Date(savedState.crawlerFinishedAt) : null;
306
- this.state.crawlerStartedAt = savedState.crawlerStartedAt ? new Date(savedState.crawlerStartedAt) : null;
307
- this.state.statsPersistedAt = savedState.statsPersistedAt ? new Date(savedState.statsPersistedAt) : null;
308
- this.state.crawlerRuntimeMillis = savedState.crawlerRuntimeMillis;
309
- this.instanceStart = Date.now() - (+this.state.statsPersistedAt - savedState.crawlerLastStartTimestamp);
310
- this.log.debug('Loaded from KeyValueStore');
311
- }
312
- teardown() {
313
- // this can be called before a call to startCapturing happens (or in a 'finally' block)
314
- // Only unsubscribe if event manager was already resolved — avoid eagerly resolving it
315
- // (e.g. during the constructor's reset() call, which would capture the wrong context)
316
- this._events?.off("persistState" /* EventType.PERSIST_STATE */, this.listener);
317
- if (this.logInterval) {
318
- clearInterval(this.logInterval);
319
- this.logInterval = null;
320
- }
321
- }
322
- /**
323
- * Make this class serializable when called with `JSON.stringify(statsInstance)` directly
324
- * or through `keyValueStore.setValue('KEY', statsInstance)`
325
- */
326
- toJSON() {
327
- // merge all the current state information that can be used from the outside
328
- // without the need to reconstruct for the sake of stats.calculate()
329
- // omit duplicated information
330
- const result = {
331
- ...this.state,
332
- crawlerLastStartTimestamp: this.instanceStart,
333
- crawlerFinishedAt: this.state.crawlerFinishedAt
334
- ? new Date(this.state.crawlerFinishedAt).toISOString()
335
- : null,
336
- crawlerStartedAt: this.state.crawlerStartedAt ? new Date(this.state.crawlerStartedAt).toISOString() : null,
337
- requestRetryHistogram: this.requestRetryHistogram,
338
- statsId: this.id,
339
- statsPersistedAt: new Date().toISOString(),
340
- ...this.calculate(),
341
- };
342
- Reflect.deleteProperty(result, 'requestsWithStatusCode');
343
- Reflect.deleteProperty(result, 'errors');
344
- Reflect.deleteProperty(result, 'retryErrors');
345
- result.requestsWithStatusCode = this.state.requestsWithStatusCode;
346
- result.errors = this.state.errors;
347
- result.retryErrors = this.state.retryErrors;
348
- return result;
349
- }
350
- }
@@ -1,264 +0,0 @@
1
- import type { BatchAddRequestsResult, Dictionary } from '@crawlee/types';
2
- import { type RobotsTxtFile } from '@crawlee/utils';
3
- import type { SetRequired } from 'type-fest';
4
- import { Request } from '../request.js';
5
- import type { IRequestManager } from '../storages/request_manager.js';
6
- import type { AddRequestsBatchedOptions, AddRequestsBatchedResult, RequestQueueOperationOptions } from '../storages/request_queue.js';
7
- import type { GlobInput, PseudoUrlInput, RegExpInput, RequestTransform, SkippedRequestCallback } from './shared.js';
8
- export interface EnqueueLinksOptions extends RequestQueueOperationOptions {
9
- /** Limit the amount of actually enqueued URLs to this number. Useful for testing across the entire crawling scope. */
10
- limit?: number;
11
- /** An array of URLs to enqueue. */
12
- urls?: readonly string[];
13
- /** A request manager to which the URLs will be enqueued. */
14
- requestManager?: IRequestManager;
15
- /** A CSS selector matching links to be enqueued. */
16
- selector?: string;
17
- /** Sets {@link Request.userData} for newly enqueued requests. */
18
- userData?: Dictionary;
19
- /**
20
- * Sets {@link Request.label} for newly enqueued requests.
21
- *
22
- * This option has the lowest priority and can be overwritten by request options
23
- * specified in `globs`, `regexps`, or `pseudoUrls` objects, as well as by `transformRequestFunction`.
24
- */
25
- label?: string;
26
- /** Sets {@link Request.sessionId} for newly enqueued requests. */
27
- sessionId?: string;
28
- /**
29
- * If set to `true`, tells the crawler to skip navigation and process the request directly.
30
- * @default false
31
- */
32
- skipNavigation?: boolean;
33
- /**
34
- * A base URL that will be used to resolve relative URLs when using Cheerio. Ignored when using Puppeteer,
35
- * since the relative URL resolution is done inside the browser automatically.
36
- */
37
- baseUrl?: string;
38
- /**
39
- * An array of glob pattern strings or plain objects
40
- * containing glob pattern strings matching the URLs to be enqueued.
41
- *
42
- * The plain objects must include at least the `glob` property, which holds the glob pattern string.
43
- * All remaining keys will be used as request options for the corresponding enqueued {@link Request} objects.
44
- *
45
- * The matching is always case-insensitive.
46
- * If you need case-sensitive matching, use `regexps` property directly.
47
- *
48
- * If `globs` is an empty array or `undefined`, and `regexps` are also not defined, then the function
49
- * enqueues the links with the same subdomain.
50
- */
51
- globs?: readonly GlobInput[];
52
- /**
53
- * An array of glob pattern strings, regexp patterns or plain objects
54
- * containing patterns matching URLs that will **never** be enqueued.
55
- *
56
- * The plain objects must include either the `glob` property or the `regexp` property.
57
- *
58
- * Glob matching is always case-insensitive.
59
- * If you need case-sensitive matching, provide a regexp.
60
- */
61
- exclude?: readonly (GlobInput | RegExpInput)[];
62
- /**
63
- * An array of regular expressions or plain objects
64
- * containing regular expressions matching the URLs to be enqueued.
65
- *
66
- * The plain objects must include at least the `regexp` property, which holds the regular expression.
67
- * All remaining keys will be used as request options for the corresponding enqueued {@link Request} objects.
68
- *
69
- * If `regexps` is an empty array or `undefined`, and `globs` are also not defined, then the function
70
- * enqueues the links with the same subdomain.
71
- */
72
- regexps?: readonly RegExpInput[];
73
- /**
74
- * *NOTE:* In future versions of SDK the options will be removed.
75
- * Please use `globs` or `regexps` instead.
76
- *
77
- * An array of {@link PseudoUrl} strings or plain objects
78
- * containing {@link PseudoUrl} strings matching the URLs to be enqueued.
79
- *
80
- * The plain objects must include at least the `purl` property, which holds the pseudo-URL string.
81
- * All remaining keys will be used as request options for the corresponding enqueued {@link Request} objects.
82
- *
83
- * With a pseudo-URL string, the matching is always case-insensitive.
84
- * If you need case-sensitive matching, use `regexps` property directly.
85
- *
86
- * If `pseudoUrls` is an empty array or `undefined`, then the function
87
- * enqueues the links with the same subdomain.
88
- *
89
- * @deprecated prefer using `globs` or `regexps` instead
90
- */
91
- pseudoUrls?: readonly PseudoUrlInput[];
92
- /**
93
- * After request options are filtered by patterns, this function can be used
94
- * to remove them or modify their contents such as `userData`, `payload` or, most importantly `uniqueKey`. This is useful
95
- * when you need to enqueue multiple `Requests` to the queue that share the same URL, but differ in methods or payloads,
96
- * or to dynamically update or create `userData`.
97
- *
98
- * For example: by adding `keepUrlFragment: true` to the request options, URL fragments will not be removed
99
- * when `uniqueKey` is computed.
100
- *
101
- * **Example:**
102
- * ```javascript
103
- * {
104
- * transformRequestFunction: (request) => {
105
- * request.userData.foo = 'bar';
106
- * request.keepUrlFragment = true;
107
- * return request;
108
- * }
109
- * }
110
- * ```
111
- *
112
- * Note that `transformRequestFunction` has the highest priority and can overwrite request options
113
- * specified in `globs`, `regexps`, or `pseudoUrls` objects, as well as the global `label` option.
114
- *
115
- * The function receives a {@link RequestOptions} object and can return either:
116
- * - The modified {@link RequestOptions} object
117
- * - `'unchanged'` to keep the original options as-is
118
- * - A falsy value or `'skip'` to exclude the request from the queue
119
- */
120
- transformRequestFunction?: RequestTransform;
121
- /**
122
- * The strategy to use when enqueueing the urls.
123
- *
124
- * Depending on the strategy you select, we will only check certain parts of the URLs found. Here is a diagram of each URL part and their name:
125
- *
126
- * ```md
127
- * Protocol Domain
128
- * ┌────┐ ┌─────────┐
129
- * https://example.crawlee.dev/...
130
- * │ └─────────────────┤
131
- * │ Hostname │
132
- * │ │
133
- * └─────────────────────────┘
134
- * Origin
135
- *```
136
- *
137
- * @default EnqueueStrategy.SameHostname
138
- */
139
- strategy?: EnqueueStrategy | 'all' | 'same-domain' | 'same-hostname' | 'same-origin';
140
- /**
141
- * By default, only the first batch (1000) of found requests will be added to the queue before resolving the call.
142
- * You can use this option to wait for adding all of them.
143
- */
144
- waitForAllRequestsToBeAdded?: boolean;
145
- /**
146
- * RobotsTxtFile instance for the current request that triggered the `enqueueLinks`.
147
- * If provided, disallowed URLs will be ignored.
148
- */
149
- robotsTxtFile?: Pick<RobotsTxtFile, 'isAllowed'>;
150
- /**
151
- * Mirrors {@link BasicCrawlerOptions.respectRobotsTxtFile}: pass `false` to disable filtering or
152
- * `{ userAgent }` to evaluate rules for a specific user-agent. Defaults to `*` when
153
- * {@link EnqueueLinksOptions.robotsTxtFile|`robotsTxtFile`} is provided.
154
- */
155
- respectRobotsTxtFile?: boolean | {
156
- userAgent?: string;
157
- };
158
- /**
159
- * When a request is skipped for some reason, you can use this callback to act on it.
160
- * This is currently fired for requests skipped
161
- * 1. based on robots.txt file,
162
- * 2. because they don't match enqueueLinks filters,
163
- * 3. or because the maxRequestsPerCrawl limit has been reached
164
- */
165
- onSkippedRequest?: SkippedRequestCallback;
166
- }
167
- /**
168
- * The different enqueueing strategies available.
169
- *
170
- * Depending on the strategy you select, we will only check certain parts of the URLs found. Here is a diagram of each URL part and their name:
171
- *
172
- * ```md
173
- * Protocol Domain
174
- * ┌────┐ ┌─────────┐
175
- * https://example.crawlee.dev/...
176
- * │ └─────────────────┤
177
- * │ Hostname │
178
- * │ │
179
- * └─────────────────────────┘
180
- * Origin
181
- *```
182
- *
183
- * - The `Protocol` is usually `http` or `https`
184
- * - The `Domain` represents the path without any possible subdomains to a website. For example, `crawlee.dev` is the domain of `https://example.crawlee.dev/`
185
- * - The `Hostname` is the full path to a website, including any subdomains. For example, `example.crawlee.dev` is the hostname of `https://example.crawlee.dev/`
186
- * - The `Origin` is the combination of the `Protocol` and `Hostname`. For example, `https://example.crawlee.dev` is the origin of `https://example.crawlee.dev/`
187
- */
188
- export declare enum EnqueueStrategy {
189
- /**
190
- * Matches any URLs found
191
- */
192
- All = "all",
193
- /**
194
- * Matches any URLs that have the same hostname.
195
- * For example, `https://wow.example.com/hello` will be matched for a base url of `https://wow.example.com/`, but
196
- * `https://example.com/hello` will not be matched.
197
- *
198
- * > This strategy will match both `http` and `https` protocols regardless of the base URL protocol.
199
- */
200
- SameHostname = "same-hostname",
201
- /**
202
- * Matches any URLs that have the same domain as the base URL.
203
- * For example, `https://wow.an.example.com` and `https://example.com` will both be matched for a base url of
204
- * `https://example.com`.
205
- *
206
- * > This strategy will match both `http` and `https` protocols regardless of the base URL protocol.
207
- */
208
- SameDomain = "same-domain",
209
- /**
210
- * Matches any URLs that have the same hostname and protocol.
211
- * For example, `https://wow.example.com/hello` will be matched for a base url of `https://wow.example.com/`, but
212
- * `http://wow.example.com/hello` will not be matched.
213
- *
214
- * > This strategy will ensure the protocol of the base URL is the same as the protocol of the URL to be enqueued.
215
- */
216
- SameOrigin = "same-origin"
217
- }
218
- /**
219
- * This function enqueues the urls provided to the {@link RequestQueue} provided. If you want to automatically find and enqueue links,
220
- * you should use the context-aware `enqueueLinks` function provided on the crawler contexts.
221
- *
222
- * Optionally, the function allows you to filter the target links' URLs using an array of globs or regular expressions
223
- * and override settings of the enqueued {@link Request} objects.
224
- *
225
- * **Example usage**
226
- *
227
- * ```javascript
228
- * await enqueueLinks({
229
- * urls: aListOfFoundUrls,
230
- * requestManager,
231
- * selector: 'a.product-detail',
232
- * globs: [
233
- * 'https://www.example.com/handbags/*',
234
- * 'https://www.example.com/purses/*'
235
- * ],
236
- * });
237
- * ```
238
- *
239
- * @param options All `enqueueLinks()` parameters are passed via an options object.
240
- * @returns Promise that resolves to {@link BatchAddRequestsResult} object.
241
- */
242
- export declare function enqueueLinks(options: SetRequired<Omit<EnqueueLinksOptions, 'requestManager'>, 'urls'> & {
243
- requestManager: {
244
- addRequestsBatched: (requests: Request<Dictionary>[], options: AddRequestsBatchedOptions) => Promise<AddRequestsBatchedResult>;
245
- };
246
- }): Promise<BatchAddRequestsResult>;
247
- /**
248
- * @internal
249
- * This method helps resolve the baseUrl that will be used for filtering in {@link enqueueLinks}.
250
- * - If a user provides a base url, we always return it
251
- * - If a user specifies {@link EnqueueStrategy.All} strategy, they do not care if the newly found urls are on the original
252
- * request domain, or a redirected one
253
- * - In all other cases, we return the domain of the original request as that's the one we need to use for filtering
254
- */
255
- export declare function resolveBaseUrlForEnqueueLinksFiltering({ enqueueStrategy, finalRequestUrl, originalRequestUrl, userProvidedBaseUrl, }: ResolveBaseUrl): string | undefined;
256
- /**
257
- * @internal
258
- */
259
- export interface ResolveBaseUrl {
260
- userProvidedBaseUrl?: string;
261
- enqueueStrategy?: EnqueueLinksOptions['strategy'];
262
- originalRequestUrl: string;
263
- finalRequestUrl?: string;
264
- }