@crawlee/core 4.0.0-beta.99 → 4.0.0-rc.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/configuration.d.ts +16 -47
- package/configuration.js +13 -25
- package/debug.js +4 -4
- package/errors.d.ts +28 -38
- package/errors.js +33 -47
- package/events/event_manager.d.ts +2 -2
- package/events/event_manager.js +7 -6
- package/events/index.d.ts +1 -0
- package/events/local_event_manager.d.ts +1 -8
- package/events/local_event_manager.js +13 -13
- package/events/system_info.d.ts +38 -0
- package/index.d.ts +2 -8
- package/index.js +4 -8
- package/internal.d.ts +8 -0
- package/internal.js +9 -0
- package/log.d.ts +10 -11
- package/log.js +52 -20
- package/memory-storage/memory-storage.d.ts +15 -18
- package/memory-storage/memory-storage.js +80 -58
- package/memory-storage/resource-clients/dataset.d.ts +1 -6
- package/memory-storage/resource-clients/dataset.js +23 -31
- package/memory-storage/resource-clients/key-value-store.d.ts +1 -10
- package/memory-storage/resource-clients/key-value-store.js +43 -67
- package/memory-storage/resource-clients/request-queue.d.ts +1 -42
- package/memory-storage/resource-clients/request-queue.js +109 -117
- package/owned_or_injected.d.ts +1 -3
- package/owned_or_injected.js +17 -17
- package/package.json +17 -20
- package/proxy_configuration.d.ts +21 -26
- package/proxy_configuration.js +35 -25
- package/recoverable_state.d.ts +104 -47
- package/recoverable_state.js +199 -74
- package/request.d.ts +20 -107
- package/request.js +78 -244
- package/serialization.js +17 -16
- package/service_locator.d.ts +22 -10
- package/service_locator.js +59 -48
- package/storages/batched_adds.d.ts +37 -0
- package/storages/batched_adds.js +73 -0
- package/storages/dataset.d.ts +13 -8
- package/storages/dataset.js +149 -40
- package/storages/index.d.ts +4 -4
- package/storages/index.js +2 -4
- package/storages/key_value_store.d.ts +16 -35
- package/storages/key_value_store.js +223 -110
- package/storages/key_value_store_codec.js +6 -11
- package/storages/request_dedup_cache.d.ts +1 -4
- package/storages/request_dedup_cache.js +15 -15
- package/storages/request_list.d.ts +9 -104
- package/storages/request_list.js +236 -233
- package/storages/request_loader.d.ts +49 -18
- package/storages/request_loader.js +36 -1
- package/storages/request_manager.d.ts +86 -0
- package/storages/request_manager_tandem.d.ts +14 -38
- package/storages/request_manager_tandem.js +67 -64
- package/storages/request_queue.d.ts +23 -50
- package/storages/request_queue.js +371 -226
- package/storages/storage_instance_manager.d.ts +2 -4
- package/storages/storage_instance_manager.js +21 -21
- package/storages/storage_stats.d.ts +1 -1
- package/storages/storage_stats.js +4 -4
- package/storages/transaction.d.ts +270 -0
- package/storages/transaction.js +296 -0
- package/storages/utils.d.ts +6 -3
- package/storages/utils.js +11 -2
- package/system-info/runtime.js +7 -7
- package/url.d.ts +9 -0
- package/url.js +11 -0
- package/validators.d.ts +23 -25
- package/validators.js +14 -25
- package/autoscaling/autoscaled_pool.d.ts +0 -213
- package/autoscaling/autoscaled_pool.js +0 -378
- package/autoscaling/client_load_signal.d.ts +0 -59
- package/autoscaling/client_load_signal.js +0 -73
- package/autoscaling/concurrency_system.d.ts +0 -283
- package/autoscaling/concurrency_system.js +0 -350
- package/autoscaling/cpu_load_signal.d.ts +0 -44
- package/autoscaling/cpu_load_signal.js +0 -46
- package/autoscaling/event_loop_load_signal.d.ts +0 -54
- package/autoscaling/event_loop_load_signal.js +0 -60
- package/autoscaling/index.d.ts +0 -9
- package/autoscaling/index.js +0 -9
- package/autoscaling/load_signal.d.ts +0 -99
- package/autoscaling/load_signal.js +0 -103
- package/autoscaling/memory_load_signal.d.ts +0 -56
- package/autoscaling/memory_load_signal.js +0 -106
- package/autoscaling/snapshotter.d.ts +0 -87
- package/autoscaling/snapshotter.js +0 -67
- package/autoscaling/system_status.d.ts +0 -161
- package/autoscaling/system_status.js +0 -139
- package/autoscaling/weighted_avg.d.ts +0 -5
- package/autoscaling/weighted_avg.js +0 -14
- package/cookie_utils.d.ts +0 -44
- package/cookie_utils.js +0 -122
- package/crawlers/context_pipeline.d.ts +0 -70
- package/crawlers/context_pipeline.js +0 -122
- package/crawlers/crawler_commons.d.ts +0 -257
- package/crawlers/crawler_commons.js +0 -107
- package/crawlers/error_snapshotter.d.ts +0 -59
- package/crawlers/error_snapshotter.js +0 -117
- package/crawlers/error_tracker.d.ts +0 -54
- package/crawlers/error_tracker.js +0 -308
- package/crawlers/index.d.ts +0 -5
- package/crawlers/index.js +0 -5
- package/crawlers/internals/types.d.ts +0 -7
- package/crawlers/statistics.d.ts +0 -209
- package/crawlers/statistics.js +0 -350
- package/enqueue_links/enqueue_links.d.ts +0 -264
- package/enqueue_links/enqueue_links.js +0 -271
- package/enqueue_links/index.d.ts +0 -2
- package/enqueue_links/index.js +0 -2
- package/enqueue_links/shared.d.ts +0 -83
- package/enqueue_links/shared.js +0 -221
- package/router.d.ts +0 -309
- package/router.js +0 -309
- package/session_pool/consts.d.ts +0 -3
- package/session_pool/consts.js +0 -3
- package/session_pool/errors.d.ts +0 -7
- package/session_pool/errors.js +0 -11
- package/session_pool/fingerprint.d.ts +0 -9
- package/session_pool/fingerprint.js +0 -30
- package/session_pool/index.d.ts +0 -4
- package/session_pool/index.js +0 -4
- package/session_pool/session.d.ts +0 -161
- package/session_pool/session.js +0 -218
- package/session_pool/session_pool.d.ts +0 -246
- package/session_pool/session_pool.js +0 -386
- package/storages/access_checking.d.ts +0 -12
- package/storages/access_checking.js +0 -17
- package/storages/sitemap_request_loader.d.ts +0 -249
- package/storages/sitemap_request_loader.js +0 -432
- /package/{crawlers/internals/types.js → events/system_info.js} +0 -0
package/crawlers/statistics.js
DELETED
|
@@ -1,350 +0,0 @@
|
|
|
1
|
-
import ow from 'ow';
|
|
2
|
-
import { serviceLocator } from '../service_locator.js';
|
|
3
|
-
import { KeyValueStore } from '../storages/key_value_store.js';
|
|
4
|
-
import { ErrorTracker } from './error_tracker.js';
|
|
5
|
-
/**
|
|
6
|
-
* @ignore
|
|
7
|
-
*/
|
|
8
|
-
class Job {
|
|
9
|
-
lastRunAt = null;
|
|
10
|
-
durationMillis;
|
|
11
|
-
run() {
|
|
12
|
-
this.lastRunAt = Date.now();
|
|
13
|
-
}
|
|
14
|
-
finish() {
|
|
15
|
-
this.durationMillis = Date.now() - this.lastRunAt;
|
|
16
|
-
return this.durationMillis;
|
|
17
|
-
}
|
|
18
|
-
}
|
|
19
|
-
const errorTrackerConfig = {
|
|
20
|
-
showErrorCode: true,
|
|
21
|
-
showErrorName: true,
|
|
22
|
-
showStackTrace: true,
|
|
23
|
-
showFullStack: false,
|
|
24
|
-
showErrorMessage: true,
|
|
25
|
-
showFullMessage: false,
|
|
26
|
-
};
|
|
27
|
-
/**
|
|
28
|
-
* The statistics class provides an interface to collecting and logging run
|
|
29
|
-
* statistics for requests.
|
|
30
|
-
*
|
|
31
|
-
* All statistic information is saved on key value store
|
|
32
|
-
* under the key `CRAWLEE_CRAWLER_STATISTICS_*`, persists between
|
|
33
|
-
* migrations and abort/resurrect
|
|
34
|
-
*
|
|
35
|
-
* @category Crawlers
|
|
36
|
-
*/
|
|
37
|
-
export class Statistics {
|
|
38
|
-
static id = 0;
|
|
39
|
-
/**
|
|
40
|
-
* An error tracker for final retry errors.
|
|
41
|
-
*/
|
|
42
|
-
errorTracker;
|
|
43
|
-
/**
|
|
44
|
-
* An error tracker for retry errors prior to the final retry.
|
|
45
|
-
*/
|
|
46
|
-
errorTrackerRetry;
|
|
47
|
-
/**
|
|
48
|
-
* Statistic instance id.
|
|
49
|
-
*/
|
|
50
|
-
id;
|
|
51
|
-
/**
|
|
52
|
-
* Current statistic state used for doing calculations on {@link Statistics.calculate} calls
|
|
53
|
-
*/
|
|
54
|
-
state;
|
|
55
|
-
/**
|
|
56
|
-
* Contains the current retries histogram. Index 0 means 0 retries, index 2, 2 retries, and so on
|
|
57
|
-
*/
|
|
58
|
-
requestRetryHistogram = [];
|
|
59
|
-
keyValueStore = undefined;
|
|
60
|
-
persistStateKey;
|
|
61
|
-
logIntervalMillis;
|
|
62
|
-
logMessage;
|
|
63
|
-
listener;
|
|
64
|
-
requestsInProgress = new Map();
|
|
65
|
-
log;
|
|
66
|
-
instanceStart;
|
|
67
|
-
logInterval;
|
|
68
|
-
_events;
|
|
69
|
-
persistenceOptions;
|
|
70
|
-
get events() {
|
|
71
|
-
if (!this._events) {
|
|
72
|
-
this._events = serviceLocator.getEventManager();
|
|
73
|
-
}
|
|
74
|
-
return this._events;
|
|
75
|
-
}
|
|
76
|
-
/**
|
|
77
|
-
* @internal
|
|
78
|
-
*/
|
|
79
|
-
constructor(options = {}) {
|
|
80
|
-
ow(options, ow.object.exactShape({
|
|
81
|
-
logIntervalSecs: ow.optional.number,
|
|
82
|
-
logMessage: ow.optional.string,
|
|
83
|
-
log: ow.optional.object,
|
|
84
|
-
keyValueStore: ow.optional.object,
|
|
85
|
-
persistenceOptions: ow.optional.object,
|
|
86
|
-
saveErrorSnapshots: ow.optional.boolean,
|
|
87
|
-
id: ow.optional.any(ow.number, ow.string),
|
|
88
|
-
}));
|
|
89
|
-
const { logIntervalSecs = 60, logMessage = 'Statistics', keyValueStore, persistenceOptions = {
|
|
90
|
-
enable: true,
|
|
91
|
-
}, saveErrorSnapshots = false, id, } = options;
|
|
92
|
-
this.id = id ?? String(Statistics.id++);
|
|
93
|
-
this.persistStateKey = `CRAWLEE_CRAWLER_STATISTICS_${this.id}`;
|
|
94
|
-
this.log = (options.log ?? serviceLocator.getLogger()).child({ prefix: 'Statistics' });
|
|
95
|
-
this.errorTracker = new ErrorTracker({ ...errorTrackerConfig, saveErrorSnapshots });
|
|
96
|
-
this.errorTrackerRetry = new ErrorTracker({ ...errorTrackerConfig, saveErrorSnapshots });
|
|
97
|
-
this.logIntervalMillis = logIntervalSecs * 1000;
|
|
98
|
-
this.logMessage = logMessage;
|
|
99
|
-
this.keyValueStore = keyValueStore;
|
|
100
|
-
this.listener = this.persistState.bind(this);
|
|
101
|
-
this.persistenceOptions = persistenceOptions;
|
|
102
|
-
// initialize by "resetting"
|
|
103
|
-
this.reset();
|
|
104
|
-
}
|
|
105
|
-
/**
|
|
106
|
-
* Set the current statistic instance to pristine values
|
|
107
|
-
*/
|
|
108
|
-
reset() {
|
|
109
|
-
this.errorTracker.reset();
|
|
110
|
-
this.errorTrackerRetry.reset();
|
|
111
|
-
this.state = {
|
|
112
|
-
requestsFinished: 0,
|
|
113
|
-
requestsFailed: 0,
|
|
114
|
-
requestsRetries: 0,
|
|
115
|
-
requestsFailedPerMinute: 0,
|
|
116
|
-
requestsFinishedPerMinute: 0,
|
|
117
|
-
requestMinDurationMillis: Infinity,
|
|
118
|
-
requestMaxDurationMillis: 0,
|
|
119
|
-
requestTotalFailedDurationMillis: 0,
|
|
120
|
-
requestTotalFinishedDurationMillis: 0,
|
|
121
|
-
crawlerStartedAt: null,
|
|
122
|
-
crawlerFinishedAt: null,
|
|
123
|
-
statsPersistedAt: null,
|
|
124
|
-
crawlerRuntimeMillis: 0,
|
|
125
|
-
requestsWithStatusCode: {},
|
|
126
|
-
errors: this.errorTracker.result,
|
|
127
|
-
retryErrors: this.errorTrackerRetry.result,
|
|
128
|
-
};
|
|
129
|
-
this.requestRetryHistogram.length = 0;
|
|
130
|
-
this.requestsInProgress.clear();
|
|
131
|
-
this.instanceStart = Date.now();
|
|
132
|
-
this.teardown();
|
|
133
|
-
}
|
|
134
|
-
/**
|
|
135
|
-
* @param options - Override the persistence options provided in the constructor
|
|
136
|
-
*/
|
|
137
|
-
async resetStore(options) {
|
|
138
|
-
if (!this.persistenceOptions.enable && !options?.enable) {
|
|
139
|
-
return;
|
|
140
|
-
}
|
|
141
|
-
if (!this.keyValueStore) {
|
|
142
|
-
return;
|
|
143
|
-
}
|
|
144
|
-
await this.keyValueStore.setValue(this.persistStateKey, null);
|
|
145
|
-
}
|
|
146
|
-
/**
|
|
147
|
-
* Increments the status code counter.
|
|
148
|
-
*/
|
|
149
|
-
registerStatusCode(code) {
|
|
150
|
-
const s = String(code);
|
|
151
|
-
if (this.state.requestsWithStatusCode[s] === undefined) {
|
|
152
|
-
this.state.requestsWithStatusCode[s] = 0;
|
|
153
|
-
}
|
|
154
|
-
this.state.requestsWithStatusCode[s]++;
|
|
155
|
-
}
|
|
156
|
-
/**
|
|
157
|
-
* Starts a job
|
|
158
|
-
* @ignore
|
|
159
|
-
*/
|
|
160
|
-
startJob(id) {
|
|
161
|
-
let job = this.requestsInProgress.get(id);
|
|
162
|
-
if (!job)
|
|
163
|
-
job = new Job();
|
|
164
|
-
job.run();
|
|
165
|
-
this.requestsInProgress.set(id, job);
|
|
166
|
-
}
|
|
167
|
-
/**
|
|
168
|
-
* Mark job as finished and sets the state
|
|
169
|
-
* @ignore
|
|
170
|
-
*/
|
|
171
|
-
finishJob(id, retryCount) {
|
|
172
|
-
const job = this.requestsInProgress.get(id);
|
|
173
|
-
if (!job)
|
|
174
|
-
return;
|
|
175
|
-
const jobDurationMillis = job.finish();
|
|
176
|
-
this.state.requestsFinished++;
|
|
177
|
-
this.state.requestTotalFinishedDurationMillis += jobDurationMillis;
|
|
178
|
-
this.saveRetryCountForJob(retryCount);
|
|
179
|
-
if (jobDurationMillis < this.state.requestMinDurationMillis)
|
|
180
|
-
this.state.requestMinDurationMillis = jobDurationMillis;
|
|
181
|
-
if (jobDurationMillis > this.state.requestMaxDurationMillis)
|
|
182
|
-
this.state.requestMaxDurationMillis = jobDurationMillis;
|
|
183
|
-
this.requestsInProgress.delete(id);
|
|
184
|
-
}
|
|
185
|
-
/**
|
|
186
|
-
* Mark job as failed and sets the state
|
|
187
|
-
* @ignore
|
|
188
|
-
*/
|
|
189
|
-
failJob(id, retryCount) {
|
|
190
|
-
const job = this.requestsInProgress.get(id);
|
|
191
|
-
if (!job)
|
|
192
|
-
return;
|
|
193
|
-
this.state.requestTotalFailedDurationMillis += job.finish();
|
|
194
|
-
this.state.requestsFailed++;
|
|
195
|
-
this.saveRetryCountForJob(retryCount);
|
|
196
|
-
this.requestsInProgress.delete(id);
|
|
197
|
-
}
|
|
198
|
-
/**
|
|
199
|
-
* Discards a started job without affecting the finished/failed counters, e.g. when a request
|
|
200
|
-
* turns out to be skipped (robots.txt, enqueue strategy) after `startJob` was already called for it.
|
|
201
|
-
* @ignore
|
|
202
|
-
*/
|
|
203
|
-
discardJob(id) {
|
|
204
|
-
this.requestsInProgress.delete(id);
|
|
205
|
-
}
|
|
206
|
-
/**
|
|
207
|
-
* Calculate the current statistics
|
|
208
|
-
*/
|
|
209
|
-
calculate() {
|
|
210
|
-
const { requestsFailed, requestsFinished, requestTotalFailedDurationMillis, requestTotalFinishedDurationMillis, } = this.state;
|
|
211
|
-
const totalMillis = Date.now() - this.instanceStart;
|
|
212
|
-
const totalMinutes = totalMillis / 1000 / 60;
|
|
213
|
-
return {
|
|
214
|
-
requestAvgFailedDurationMillis: Math.round(requestTotalFailedDurationMillis / requestsFailed) || Infinity,
|
|
215
|
-
requestAvgFinishedDurationMillis: Math.round(requestTotalFinishedDurationMillis / requestsFinished) || Infinity,
|
|
216
|
-
requestsFinishedPerMinute: Math.round(requestsFinished / totalMinutes) || 0,
|
|
217
|
-
requestsFailedPerMinute: Math.floor(requestsFailed / totalMinutes) || 0,
|
|
218
|
-
requestTotalDurationMillis: requestTotalFinishedDurationMillis + requestTotalFailedDurationMillis,
|
|
219
|
-
requestsTotal: requestsFailed + requestsFinished,
|
|
220
|
-
crawlerRuntimeMillis: totalMillis,
|
|
221
|
-
};
|
|
222
|
-
}
|
|
223
|
-
/**
|
|
224
|
-
* Initializes the key value store for persisting the statistics,
|
|
225
|
-
* displaying the current state in predefined intervals
|
|
226
|
-
*/
|
|
227
|
-
async startCapturing() {
|
|
228
|
-
this.keyValueStore ??= await KeyValueStore.open(null, { configuration: serviceLocator.getConfiguration() });
|
|
229
|
-
if (this.state.crawlerStartedAt === null) {
|
|
230
|
-
this.state.crawlerStartedAt = new Date();
|
|
231
|
-
}
|
|
232
|
-
if (this.persistenceOptions.enable) {
|
|
233
|
-
await this.maybeLoadStatistics();
|
|
234
|
-
this.events.on("persistState" /* EventType.PERSIST_STATE */, this.listener);
|
|
235
|
-
}
|
|
236
|
-
this.logInterval = setInterval(() => {
|
|
237
|
-
this.log.info(this.logMessage, {
|
|
238
|
-
...this.calculate(),
|
|
239
|
-
retryHistogram: this.requestRetryHistogram,
|
|
240
|
-
});
|
|
241
|
-
}, this.logIntervalMillis);
|
|
242
|
-
}
|
|
243
|
-
/**
|
|
244
|
-
* Stops logging and remove event listeners, then persist
|
|
245
|
-
*/
|
|
246
|
-
async stopCapturing() {
|
|
247
|
-
this.teardown();
|
|
248
|
-
this.state.crawlerFinishedAt = new Date();
|
|
249
|
-
await this.persistState();
|
|
250
|
-
}
|
|
251
|
-
saveRetryCountForJob(retryCount) {
|
|
252
|
-
if (retryCount > 0)
|
|
253
|
-
this.state.requestsRetries++;
|
|
254
|
-
this.requestRetryHistogram[retryCount] ??= 0;
|
|
255
|
-
this.requestRetryHistogram[retryCount]++;
|
|
256
|
-
}
|
|
257
|
-
/**
|
|
258
|
-
* Persist internal state to the key value store
|
|
259
|
-
* @param options - Override the persistence options provided in the constructor
|
|
260
|
-
*/
|
|
261
|
-
async persistState(options) {
|
|
262
|
-
if (!this.persistenceOptions.enable && !options?.enable) {
|
|
263
|
-
return;
|
|
264
|
-
}
|
|
265
|
-
// this might be called before startCapturing was called without using await, should not crash
|
|
266
|
-
if (!this.keyValueStore) {
|
|
267
|
-
return;
|
|
268
|
-
}
|
|
269
|
-
this.log.debug('Persisting state', { persistStateKey: this.persistStateKey });
|
|
270
|
-
await this.keyValueStore
|
|
271
|
-
.setValue(this.persistStateKey, this.toJSON())
|
|
272
|
-
.catch((error) => this.log.warning(`Failed to persist the statistics to ${this.persistStateKey}`, { error }));
|
|
273
|
-
}
|
|
274
|
-
/**
|
|
275
|
-
* Loads the current statistic from the key value store if any
|
|
276
|
-
*/
|
|
277
|
-
async maybeLoadStatistics() {
|
|
278
|
-
// this might be called before startCapturing was called without using await, should not crash
|
|
279
|
-
if (!this.keyValueStore) {
|
|
280
|
-
return;
|
|
281
|
-
}
|
|
282
|
-
const savedState = await this.keyValueStore.getValue(this.persistStateKey);
|
|
283
|
-
if (!savedState)
|
|
284
|
-
return;
|
|
285
|
-
// We saw a run where the requestRetryHistogram was not iterable and crashed
|
|
286
|
-
// the crawler. Adding some logging to monitor this problem in the future.
|
|
287
|
-
if (!Array.isArray(savedState.requestRetryHistogram)) {
|
|
288
|
-
this.log.warning('Received invalid state from Key-value store.', {
|
|
289
|
-
persistStateKey: this.persistStateKey,
|
|
290
|
-
state: savedState,
|
|
291
|
-
});
|
|
292
|
-
}
|
|
293
|
-
this.log.debug('Recreating state from KeyValueStore', { persistStateKey: this.persistStateKey });
|
|
294
|
-
// the `requestRetryHistogram` array might be very large, we could end up with
|
|
295
|
-
// `RangeError: Maximum call stack size exceeded` if we use `a.push(...b)`
|
|
296
|
-
savedState.requestRetryHistogram.forEach((idx) => this.requestRetryHistogram.push(idx));
|
|
297
|
-
this.state.requestsFinished = savedState.requestsFinished;
|
|
298
|
-
this.state.requestsFailed = savedState.requestsFailed;
|
|
299
|
-
this.state.requestsRetries = savedState.requestsRetries;
|
|
300
|
-
this.state.requestTotalFailedDurationMillis = savedState.requestTotalFailedDurationMillis;
|
|
301
|
-
this.state.requestTotalFinishedDurationMillis = savedState.requestTotalFinishedDurationMillis;
|
|
302
|
-
this.state.requestMinDurationMillis = savedState.requestMinDurationMillis;
|
|
303
|
-
this.state.requestMaxDurationMillis = savedState.requestMaxDurationMillis;
|
|
304
|
-
// persisted state uses ISO date strings
|
|
305
|
-
this.state.crawlerFinishedAt = savedState.crawlerFinishedAt ? new Date(savedState.crawlerFinishedAt) : null;
|
|
306
|
-
this.state.crawlerStartedAt = savedState.crawlerStartedAt ? new Date(savedState.crawlerStartedAt) : null;
|
|
307
|
-
this.state.statsPersistedAt = savedState.statsPersistedAt ? new Date(savedState.statsPersistedAt) : null;
|
|
308
|
-
this.state.crawlerRuntimeMillis = savedState.crawlerRuntimeMillis;
|
|
309
|
-
this.instanceStart = Date.now() - (+this.state.statsPersistedAt - savedState.crawlerLastStartTimestamp);
|
|
310
|
-
this.log.debug('Loaded from KeyValueStore');
|
|
311
|
-
}
|
|
312
|
-
teardown() {
|
|
313
|
-
// this can be called before a call to startCapturing happens (or in a 'finally' block)
|
|
314
|
-
// Only unsubscribe if event manager was already resolved — avoid eagerly resolving it
|
|
315
|
-
// (e.g. during the constructor's reset() call, which would capture the wrong context)
|
|
316
|
-
this._events?.off("persistState" /* EventType.PERSIST_STATE */, this.listener);
|
|
317
|
-
if (this.logInterval) {
|
|
318
|
-
clearInterval(this.logInterval);
|
|
319
|
-
this.logInterval = null;
|
|
320
|
-
}
|
|
321
|
-
}
|
|
322
|
-
/**
|
|
323
|
-
* Make this class serializable when called with `JSON.stringify(statsInstance)` directly
|
|
324
|
-
* or through `keyValueStore.setValue('KEY', statsInstance)`
|
|
325
|
-
*/
|
|
326
|
-
toJSON() {
|
|
327
|
-
// merge all the current state information that can be used from the outside
|
|
328
|
-
// without the need to reconstruct for the sake of stats.calculate()
|
|
329
|
-
// omit duplicated information
|
|
330
|
-
const result = {
|
|
331
|
-
...this.state,
|
|
332
|
-
crawlerLastStartTimestamp: this.instanceStart,
|
|
333
|
-
crawlerFinishedAt: this.state.crawlerFinishedAt
|
|
334
|
-
? new Date(this.state.crawlerFinishedAt).toISOString()
|
|
335
|
-
: null,
|
|
336
|
-
crawlerStartedAt: this.state.crawlerStartedAt ? new Date(this.state.crawlerStartedAt).toISOString() : null,
|
|
337
|
-
requestRetryHistogram: this.requestRetryHistogram,
|
|
338
|
-
statsId: this.id,
|
|
339
|
-
statsPersistedAt: new Date().toISOString(),
|
|
340
|
-
...this.calculate(),
|
|
341
|
-
};
|
|
342
|
-
Reflect.deleteProperty(result, 'requestsWithStatusCode');
|
|
343
|
-
Reflect.deleteProperty(result, 'errors');
|
|
344
|
-
Reflect.deleteProperty(result, 'retryErrors');
|
|
345
|
-
result.requestsWithStatusCode = this.state.requestsWithStatusCode;
|
|
346
|
-
result.errors = this.state.errors;
|
|
347
|
-
result.retryErrors = this.state.retryErrors;
|
|
348
|
-
return result;
|
|
349
|
-
}
|
|
350
|
-
}
|
|
@@ -1,264 +0,0 @@
|
|
|
1
|
-
import type { BatchAddRequestsResult, Dictionary } from '@crawlee/types';
|
|
2
|
-
import { type RobotsTxtFile } from '@crawlee/utils';
|
|
3
|
-
import type { SetRequired } from 'type-fest';
|
|
4
|
-
import { Request } from '../request.js';
|
|
5
|
-
import type { IRequestManager } from '../storages/request_manager.js';
|
|
6
|
-
import type { AddRequestsBatchedOptions, AddRequestsBatchedResult, RequestQueueOperationOptions } from '../storages/request_queue.js';
|
|
7
|
-
import type { GlobInput, PseudoUrlInput, RegExpInput, RequestTransform, SkippedRequestCallback } from './shared.js';
|
|
8
|
-
export interface EnqueueLinksOptions extends RequestQueueOperationOptions {
|
|
9
|
-
/** Limit the amount of actually enqueued URLs to this number. Useful for testing across the entire crawling scope. */
|
|
10
|
-
limit?: number;
|
|
11
|
-
/** An array of URLs to enqueue. */
|
|
12
|
-
urls?: readonly string[];
|
|
13
|
-
/** A request manager to which the URLs will be enqueued. */
|
|
14
|
-
requestManager?: IRequestManager;
|
|
15
|
-
/** A CSS selector matching links to be enqueued. */
|
|
16
|
-
selector?: string;
|
|
17
|
-
/** Sets {@link Request.userData} for newly enqueued requests. */
|
|
18
|
-
userData?: Dictionary;
|
|
19
|
-
/**
|
|
20
|
-
* Sets {@link Request.label} for newly enqueued requests.
|
|
21
|
-
*
|
|
22
|
-
* This option has the lowest priority and can be overwritten by request options
|
|
23
|
-
* specified in `globs`, `regexps`, or `pseudoUrls` objects, as well as by `transformRequestFunction`.
|
|
24
|
-
*/
|
|
25
|
-
label?: string;
|
|
26
|
-
/** Sets {@link Request.sessionId} for newly enqueued requests. */
|
|
27
|
-
sessionId?: string;
|
|
28
|
-
/**
|
|
29
|
-
* If set to `true`, tells the crawler to skip navigation and process the request directly.
|
|
30
|
-
* @default false
|
|
31
|
-
*/
|
|
32
|
-
skipNavigation?: boolean;
|
|
33
|
-
/**
|
|
34
|
-
* A base URL that will be used to resolve relative URLs when using Cheerio. Ignored when using Puppeteer,
|
|
35
|
-
* since the relative URL resolution is done inside the browser automatically.
|
|
36
|
-
*/
|
|
37
|
-
baseUrl?: string;
|
|
38
|
-
/**
|
|
39
|
-
* An array of glob pattern strings or plain objects
|
|
40
|
-
* containing glob pattern strings matching the URLs to be enqueued.
|
|
41
|
-
*
|
|
42
|
-
* The plain objects must include at least the `glob` property, which holds the glob pattern string.
|
|
43
|
-
* All remaining keys will be used as request options for the corresponding enqueued {@link Request} objects.
|
|
44
|
-
*
|
|
45
|
-
* The matching is always case-insensitive.
|
|
46
|
-
* If you need case-sensitive matching, use `regexps` property directly.
|
|
47
|
-
*
|
|
48
|
-
* If `globs` is an empty array or `undefined`, and `regexps` are also not defined, then the function
|
|
49
|
-
* enqueues the links with the same subdomain.
|
|
50
|
-
*/
|
|
51
|
-
globs?: readonly GlobInput[];
|
|
52
|
-
/**
|
|
53
|
-
* An array of glob pattern strings, regexp patterns or plain objects
|
|
54
|
-
* containing patterns matching URLs that will **never** be enqueued.
|
|
55
|
-
*
|
|
56
|
-
* The plain objects must include either the `glob` property or the `regexp` property.
|
|
57
|
-
*
|
|
58
|
-
* Glob matching is always case-insensitive.
|
|
59
|
-
* If you need case-sensitive matching, provide a regexp.
|
|
60
|
-
*/
|
|
61
|
-
exclude?: readonly (GlobInput | RegExpInput)[];
|
|
62
|
-
/**
|
|
63
|
-
* An array of regular expressions or plain objects
|
|
64
|
-
* containing regular expressions matching the URLs to be enqueued.
|
|
65
|
-
*
|
|
66
|
-
* The plain objects must include at least the `regexp` property, which holds the regular expression.
|
|
67
|
-
* All remaining keys will be used as request options for the corresponding enqueued {@link Request} objects.
|
|
68
|
-
*
|
|
69
|
-
* If `regexps` is an empty array or `undefined`, and `globs` are also not defined, then the function
|
|
70
|
-
* enqueues the links with the same subdomain.
|
|
71
|
-
*/
|
|
72
|
-
regexps?: readonly RegExpInput[];
|
|
73
|
-
/**
|
|
74
|
-
* *NOTE:* In future versions of SDK the options will be removed.
|
|
75
|
-
* Please use `globs` or `regexps` instead.
|
|
76
|
-
*
|
|
77
|
-
* An array of {@link PseudoUrl} strings or plain objects
|
|
78
|
-
* containing {@link PseudoUrl} strings matching the URLs to be enqueued.
|
|
79
|
-
*
|
|
80
|
-
* The plain objects must include at least the `purl` property, which holds the pseudo-URL string.
|
|
81
|
-
* All remaining keys will be used as request options for the corresponding enqueued {@link Request} objects.
|
|
82
|
-
*
|
|
83
|
-
* With a pseudo-URL string, the matching is always case-insensitive.
|
|
84
|
-
* If you need case-sensitive matching, use `regexps` property directly.
|
|
85
|
-
*
|
|
86
|
-
* If `pseudoUrls` is an empty array or `undefined`, then the function
|
|
87
|
-
* enqueues the links with the same subdomain.
|
|
88
|
-
*
|
|
89
|
-
* @deprecated prefer using `globs` or `regexps` instead
|
|
90
|
-
*/
|
|
91
|
-
pseudoUrls?: readonly PseudoUrlInput[];
|
|
92
|
-
/**
|
|
93
|
-
* After request options are filtered by patterns, this function can be used
|
|
94
|
-
* to remove them or modify their contents such as `userData`, `payload` or, most importantly `uniqueKey`. This is useful
|
|
95
|
-
* when you need to enqueue multiple `Requests` to the queue that share the same URL, but differ in methods or payloads,
|
|
96
|
-
* or to dynamically update or create `userData`.
|
|
97
|
-
*
|
|
98
|
-
* For example: by adding `keepUrlFragment: true` to the request options, URL fragments will not be removed
|
|
99
|
-
* when `uniqueKey` is computed.
|
|
100
|
-
*
|
|
101
|
-
* **Example:**
|
|
102
|
-
* ```javascript
|
|
103
|
-
* {
|
|
104
|
-
* transformRequestFunction: (request) => {
|
|
105
|
-
* request.userData.foo = 'bar';
|
|
106
|
-
* request.keepUrlFragment = true;
|
|
107
|
-
* return request;
|
|
108
|
-
* }
|
|
109
|
-
* }
|
|
110
|
-
* ```
|
|
111
|
-
*
|
|
112
|
-
* Note that `transformRequestFunction` has the highest priority and can overwrite request options
|
|
113
|
-
* specified in `globs`, `regexps`, or `pseudoUrls` objects, as well as the global `label` option.
|
|
114
|
-
*
|
|
115
|
-
* The function receives a {@link RequestOptions} object and can return either:
|
|
116
|
-
* - The modified {@link RequestOptions} object
|
|
117
|
-
* - `'unchanged'` to keep the original options as-is
|
|
118
|
-
* - A falsy value or `'skip'` to exclude the request from the queue
|
|
119
|
-
*/
|
|
120
|
-
transformRequestFunction?: RequestTransform;
|
|
121
|
-
/**
|
|
122
|
-
* The strategy to use when enqueueing the urls.
|
|
123
|
-
*
|
|
124
|
-
* Depending on the strategy you select, we will only check certain parts of the URLs found. Here is a diagram of each URL part and their name:
|
|
125
|
-
*
|
|
126
|
-
* ```md
|
|
127
|
-
* Protocol Domain
|
|
128
|
-
* ┌────┐ ┌─────────┐
|
|
129
|
-
* https://example.crawlee.dev/...
|
|
130
|
-
* │ └─────────────────┤
|
|
131
|
-
* │ Hostname │
|
|
132
|
-
* │ │
|
|
133
|
-
* └─────────────────────────┘
|
|
134
|
-
* Origin
|
|
135
|
-
*```
|
|
136
|
-
*
|
|
137
|
-
* @default EnqueueStrategy.SameHostname
|
|
138
|
-
*/
|
|
139
|
-
strategy?: EnqueueStrategy | 'all' | 'same-domain' | 'same-hostname' | 'same-origin';
|
|
140
|
-
/**
|
|
141
|
-
* By default, only the first batch (1000) of found requests will be added to the queue before resolving the call.
|
|
142
|
-
* You can use this option to wait for adding all of them.
|
|
143
|
-
*/
|
|
144
|
-
waitForAllRequestsToBeAdded?: boolean;
|
|
145
|
-
/**
|
|
146
|
-
* RobotsTxtFile instance for the current request that triggered the `enqueueLinks`.
|
|
147
|
-
* If provided, disallowed URLs will be ignored.
|
|
148
|
-
*/
|
|
149
|
-
robotsTxtFile?: Pick<RobotsTxtFile, 'isAllowed'>;
|
|
150
|
-
/**
|
|
151
|
-
* Mirrors {@link BasicCrawlerOptions.respectRobotsTxtFile}: pass `false` to disable filtering or
|
|
152
|
-
* `{ userAgent }` to evaluate rules for a specific user-agent. Defaults to `*` when
|
|
153
|
-
* {@link EnqueueLinksOptions.robotsTxtFile|`robotsTxtFile`} is provided.
|
|
154
|
-
*/
|
|
155
|
-
respectRobotsTxtFile?: boolean | {
|
|
156
|
-
userAgent?: string;
|
|
157
|
-
};
|
|
158
|
-
/**
|
|
159
|
-
* When a request is skipped for some reason, you can use this callback to act on it.
|
|
160
|
-
* This is currently fired for requests skipped
|
|
161
|
-
* 1. based on robots.txt file,
|
|
162
|
-
* 2. because they don't match enqueueLinks filters,
|
|
163
|
-
* 3. or because the maxRequestsPerCrawl limit has been reached
|
|
164
|
-
*/
|
|
165
|
-
onSkippedRequest?: SkippedRequestCallback;
|
|
166
|
-
}
|
|
167
|
-
/**
|
|
168
|
-
* The different enqueueing strategies available.
|
|
169
|
-
*
|
|
170
|
-
* Depending on the strategy you select, we will only check certain parts of the URLs found. Here is a diagram of each URL part and their name:
|
|
171
|
-
*
|
|
172
|
-
* ```md
|
|
173
|
-
* Protocol Domain
|
|
174
|
-
* ┌────┐ ┌─────────┐
|
|
175
|
-
* https://example.crawlee.dev/...
|
|
176
|
-
* │ └─────────────────┤
|
|
177
|
-
* │ Hostname │
|
|
178
|
-
* │ │
|
|
179
|
-
* └─────────────────────────┘
|
|
180
|
-
* Origin
|
|
181
|
-
*```
|
|
182
|
-
*
|
|
183
|
-
* - The `Protocol` is usually `http` or `https`
|
|
184
|
-
* - The `Domain` represents the path without any possible subdomains to a website. For example, `crawlee.dev` is the domain of `https://example.crawlee.dev/`
|
|
185
|
-
* - The `Hostname` is the full path to a website, including any subdomains. For example, `example.crawlee.dev` is the hostname of `https://example.crawlee.dev/`
|
|
186
|
-
* - The `Origin` is the combination of the `Protocol` and `Hostname`. For example, `https://example.crawlee.dev` is the origin of `https://example.crawlee.dev/`
|
|
187
|
-
*/
|
|
188
|
-
export declare enum EnqueueStrategy {
|
|
189
|
-
/**
|
|
190
|
-
* Matches any URLs found
|
|
191
|
-
*/
|
|
192
|
-
All = "all",
|
|
193
|
-
/**
|
|
194
|
-
* Matches any URLs that have the same hostname.
|
|
195
|
-
* For example, `https://wow.example.com/hello` will be matched for a base url of `https://wow.example.com/`, but
|
|
196
|
-
* `https://example.com/hello` will not be matched.
|
|
197
|
-
*
|
|
198
|
-
* > This strategy will match both `http` and `https` protocols regardless of the base URL protocol.
|
|
199
|
-
*/
|
|
200
|
-
SameHostname = "same-hostname",
|
|
201
|
-
/**
|
|
202
|
-
* Matches any URLs that have the same domain as the base URL.
|
|
203
|
-
* For example, `https://wow.an.example.com` and `https://example.com` will both be matched for a base url of
|
|
204
|
-
* `https://example.com`.
|
|
205
|
-
*
|
|
206
|
-
* > This strategy will match both `http` and `https` protocols regardless of the base URL protocol.
|
|
207
|
-
*/
|
|
208
|
-
SameDomain = "same-domain",
|
|
209
|
-
/**
|
|
210
|
-
* Matches any URLs that have the same hostname and protocol.
|
|
211
|
-
* For example, `https://wow.example.com/hello` will be matched for a base url of `https://wow.example.com/`, but
|
|
212
|
-
* `http://wow.example.com/hello` will not be matched.
|
|
213
|
-
*
|
|
214
|
-
* > This strategy will ensure the protocol of the base URL is the same as the protocol of the URL to be enqueued.
|
|
215
|
-
*/
|
|
216
|
-
SameOrigin = "same-origin"
|
|
217
|
-
}
|
|
218
|
-
/**
|
|
219
|
-
* This function enqueues the urls provided to the {@link RequestQueue} provided. If you want to automatically find and enqueue links,
|
|
220
|
-
* you should use the context-aware `enqueueLinks` function provided on the crawler contexts.
|
|
221
|
-
*
|
|
222
|
-
* Optionally, the function allows you to filter the target links' URLs using an array of globs or regular expressions
|
|
223
|
-
* and override settings of the enqueued {@link Request} objects.
|
|
224
|
-
*
|
|
225
|
-
* **Example usage**
|
|
226
|
-
*
|
|
227
|
-
* ```javascript
|
|
228
|
-
* await enqueueLinks({
|
|
229
|
-
* urls: aListOfFoundUrls,
|
|
230
|
-
* requestManager,
|
|
231
|
-
* selector: 'a.product-detail',
|
|
232
|
-
* globs: [
|
|
233
|
-
* 'https://www.example.com/handbags/*',
|
|
234
|
-
* 'https://www.example.com/purses/*'
|
|
235
|
-
* ],
|
|
236
|
-
* });
|
|
237
|
-
* ```
|
|
238
|
-
*
|
|
239
|
-
* @param options All `enqueueLinks()` parameters are passed via an options object.
|
|
240
|
-
* @returns Promise that resolves to {@link BatchAddRequestsResult} object.
|
|
241
|
-
*/
|
|
242
|
-
export declare function enqueueLinks(options: SetRequired<Omit<EnqueueLinksOptions, 'requestManager'>, 'urls'> & {
|
|
243
|
-
requestManager: {
|
|
244
|
-
addRequestsBatched: (requests: Request<Dictionary>[], options: AddRequestsBatchedOptions) => Promise<AddRequestsBatchedResult>;
|
|
245
|
-
};
|
|
246
|
-
}): Promise<BatchAddRequestsResult>;
|
|
247
|
-
/**
|
|
248
|
-
* @internal
|
|
249
|
-
* This method helps resolve the baseUrl that will be used for filtering in {@link enqueueLinks}.
|
|
250
|
-
* - If a user provides a base url, we always return it
|
|
251
|
-
* - If a user specifies {@link EnqueueStrategy.All} strategy, they do not care if the newly found urls are on the original
|
|
252
|
-
* request domain, or a redirected one
|
|
253
|
-
* - In all other cases, we return the domain of the original request as that's the one we need to use for filtering
|
|
254
|
-
*/
|
|
255
|
-
export declare function resolveBaseUrlForEnqueueLinksFiltering({ enqueueStrategy, finalRequestUrl, originalRequestUrl, userProvidedBaseUrl, }: ResolveBaseUrl): string | undefined;
|
|
256
|
-
/**
|
|
257
|
-
* @internal
|
|
258
|
-
*/
|
|
259
|
-
export interface ResolveBaseUrl {
|
|
260
|
-
userProvidedBaseUrl?: string;
|
|
261
|
-
enqueueStrategy?: EnqueueLinksOptions['strategy'];
|
|
262
|
-
originalRequestUrl: string;
|
|
263
|
-
finalRequestUrl?: string;
|
|
264
|
-
}
|