@crawlee/core 4.0.0-rc.0 → 4.0.0-rc.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/configuration.d.ts +15 -46
- package/configuration.js +8 -20
- package/errors.d.ts +12 -53
- package/errors.js +13 -66
- package/events/index.d.ts +1 -0
- package/events/local_event_manager.d.ts +0 -7
- package/events/local_event_manager.js +8 -8
- package/events/system_info.d.ts +38 -0
- package/index.d.ts +2 -8
- package/index.js +4 -8
- package/internal.d.ts +8 -0
- package/internal.js +9 -0
- package/log.d.ts +10 -11
- package/log.js +53 -25
- package/memory-storage/memory-storage.d.ts +12 -7
- package/memory-storage/memory-storage.js +45 -17
- package/memory-storage/resource-clients/dataset.d.ts +0 -5
- package/memory-storage/resource-clients/dataset.js +17 -20
- package/memory-storage/resource-clients/key-value-store.d.ts +0 -9
- package/memory-storage/resource-clients/key-value-store.js +11 -33
- package/memory-storage/resource-clients/request-queue.d.ts +0 -22
- package/memory-storage/resource-clients/request-queue.js +57 -53
- package/package.json +16 -18
- package/proxy_configuration.d.ts +20 -23
- package/proxy_configuration.js +18 -12
- package/recoverable_state.d.ts +28 -6
- package/recoverable_state.js +51 -14
- package/request.d.ts +18 -104
- package/request.js +41 -220
- package/serialization.js +3 -3
- package/service_locator.d.ts +3 -0
- package/service_locator.js +2 -0
- package/storages/dataset.d.ts +9 -15
- package/storages/dataset.js +22 -14
- package/storages/index.d.ts +3 -4
- package/storages/index.js +1 -4
- package/storages/key_value_store.d.ts +12 -46
- package/storages/key_value_store.js +31 -47
- package/storages/key_value_store_codec.js +6 -11
- package/storages/request_dedup_cache.d.ts +0 -2
- package/storages/request_dedup_cache.js +8 -8
- package/storages/request_list.d.ts +6 -82
- package/storages/request_list.js +175 -179
- package/storages/request_loader.d.ts +48 -22
- package/storages/request_loader.js +36 -1
- package/storages/request_manager.d.ts +86 -0
- package/storages/request_manager_tandem.d.ts +13 -28
- package/storages/request_manager_tandem.js +46 -43
- package/storages/request_queue.d.ts +19 -49
- package/storages/request_queue.js +92 -88
- package/storages/storage_instance_manager.d.ts +1 -2
- package/storages/storage_instance_manager.js +4 -4
- package/storages/transaction.d.ts +27 -9
- package/storages/transaction.js +56 -11
- package/storages/utils.d.ts +2 -2
- package/validators.d.ts +3 -2
- package/validators.js +3 -2
- package/autoscaling/autoscaled_pool.d.ts +0 -195
- package/autoscaling/autoscaled_pool.js +0 -386
- package/autoscaling/concurrency_system.d.ts +0 -268
- package/autoscaling/concurrency_system.js +0 -362
- package/autoscaling/cpu_load_signal.d.ts +0 -43
- package/autoscaling/cpu_load_signal.js +0 -47
- package/autoscaling/event_loop_load_signal.d.ts +0 -51
- package/autoscaling/event_loop_load_signal.js +0 -60
- package/autoscaling/index.d.ts +0 -9
- package/autoscaling/index.js +0 -9
- package/autoscaling/load_signal.d.ts +0 -100
- package/autoscaling/load_signal.js +0 -105
- package/autoscaling/memory_load_signal.d.ts +0 -47
- package/autoscaling/memory_load_signal.js +0 -106
- package/autoscaling/snapshotter.d.ts +0 -84
- package/autoscaling/snapshotter.js +0 -67
- package/autoscaling/storage_backend_load_signal.d.ts +0 -56
- package/autoscaling/storage_backend_load_signal.js +0 -73
- package/autoscaling/system_status.d.ts +0 -159
- package/autoscaling/system_status.js +0 -139
- package/autoscaling/weighted_avg.d.ts +0 -5
- package/autoscaling/weighted_avg.js +0 -14
- package/cookie_utils.d.ts +0 -44
- package/cookie_utils.js +0 -122
- package/crawlers/context_pipeline.d.ts +0 -70
- package/crawlers/context_pipeline.js +0 -122
- package/crawlers/crawler_commons.d.ts +0 -159
- package/crawlers/error_snapshotter.d.ts +0 -57
- package/crawlers/error_snapshotter.js +0 -117
- package/crawlers/error_tracker.d.ts +0 -54
- package/crawlers/error_tracker.js +0 -308
- package/crawlers/index.d.ts +0 -5
- package/crawlers/index.js +0 -4
- package/crawlers/internals/types.d.ts +0 -7
- package/crawlers/internals/types.js +0 -1
- package/crawlers/statistics.d.ts +0 -328
- package/crawlers/statistics.js +0 -536
- package/enqueue_links/enqueue_links.d.ts +0 -156
- package/enqueue_links/enqueue_links.js +0 -78
- package/enqueue_links/index.d.ts +0 -2
- package/enqueue_links/index.js +0 -2
- package/enqueue_links/shared.d.ts +0 -93
- package/enqueue_links/shared.js +0 -239
- package/http.d.ts +0 -9
- package/http.js +0 -28
- package/router.d.ts +0 -306
- package/router.js +0 -309
- package/session_pool/consts.d.ts +0 -3
- package/session_pool/consts.js +0 -3
- package/session_pool/errors.d.ts +0 -7
- package/session_pool/errors.js +0 -11
- package/session_pool/fingerprint.d.ts +0 -9
- package/session_pool/fingerprint.js +0 -30
- package/session_pool/index.d.ts +0 -4
- package/session_pool/index.js +0 -4
- package/session_pool/session.d.ts +0 -150
- package/session_pool/session.js +0 -220
- package/session_pool/session_pool.d.ts +0 -240
- package/session_pool/session_pool.js +0 -394
- package/storages/sitemap_request_loader.d.ts +0 -201
- package/storages/sitemap_request_loader.js +0 -438
- package/storages/throttling_request_manager.d.ts +0 -239
- package/storages/throttling_request_manager.js +0 -646
- /package/{crawlers/crawler_commons.js → events/system_info.js} +0 -0
|
@@ -1,438 +0,0 @@
|
|
|
1
|
-
import { Transform } from 'node:stream';
|
|
2
|
-
import { EnqueueStrategy, parseSitemap } from '@crawlee/utils';
|
|
3
|
-
import { minimatch } from 'minimatch';
|
|
4
|
-
import { z } from 'zod';
|
|
5
|
-
import { constructUrlPatternObjects, urlPatternSchema } from '../enqueue_links/shared.js';
|
|
6
|
-
import { EventType } from '../events/event_manager.js';
|
|
7
|
-
import { Request } from '../request.js';
|
|
8
|
-
import { serviceLocator } from '../service_locator.js';
|
|
9
|
-
import { parseArgument, schemas } from '../validators.js';
|
|
10
|
-
import { KeyValueStore } from './key_value_store.js';
|
|
11
|
-
import { purgeDefaultStorages } from './utils.js';
|
|
12
|
-
const sitemapRequestLoaderOptionsSchema = z.strictObject({
|
|
13
|
-
sitemapUrls: schemas.arrayOf(z.string(), 'strings'),
|
|
14
|
-
proxyUrl: z.string().optional(),
|
|
15
|
-
persistStateKey: z.string().optional(),
|
|
16
|
-
signal: z.unknown().optional(),
|
|
17
|
-
timeoutMillis: schemas.anyNumber.optional(),
|
|
18
|
-
maxBufferSize: schemas.anyNumber.default(200),
|
|
19
|
-
enqueueStrategy: z.enum(EnqueueStrategy).default(EnqueueStrategy.SameHostname),
|
|
20
|
-
parseSitemapOptions: z.looseObject({}).optional(),
|
|
21
|
-
include: schemas.arrayOf(urlPatternSchema, 'URL patterns').optional(),
|
|
22
|
-
exclude: schemas.arrayOf(urlPatternSchema, 'URL patterns').optional(),
|
|
23
|
-
persistenceOptions: z.looseObject({}).optional(),
|
|
24
|
-
httpClient: schemas.httpClient.optional(),
|
|
25
|
-
});
|
|
26
|
-
/** @internal */
|
|
27
|
-
const STATE_PERSISTENCE_KEY = 'SITEMAP_REQUEST_LOADER_STATE';
|
|
28
|
-
/**
|
|
29
|
-
* A list of URLs to crawl parsed from a sitemap.
|
|
30
|
-
*
|
|
31
|
-
* The loading of the sitemap is performed in the background so that crawling can start before the sitemap is fully loaded.
|
|
32
|
-
*/
|
|
33
|
-
export class SitemapRequestLoader {
|
|
34
|
-
/**
|
|
35
|
-
* Set of URLs that were returned by `fetchNextRequest()` and not marked as handled yet.
|
|
36
|
-
* @internal
|
|
37
|
-
*/
|
|
38
|
-
inProgress = new Set();
|
|
39
|
-
/**
|
|
40
|
-
* Map of returned Request objects that have not been marked as handled yet.
|
|
41
|
-
*
|
|
42
|
-
* We use this to persist custom user fields on the in-progress requests.
|
|
43
|
-
*/
|
|
44
|
-
#requestData = new Map();
|
|
45
|
-
/**
|
|
46
|
-
* Object for keeping track of the sitemap parsing progress.
|
|
47
|
-
*/
|
|
48
|
-
#sitemapParsingProgress = {
|
|
49
|
-
/**
|
|
50
|
-
* URL of the sitemap that is currently being parsed. `null` if no sitemap is being parsed.
|
|
51
|
-
*/
|
|
52
|
-
inProgressSitemapUrl: null,
|
|
53
|
-
/**
|
|
54
|
-
* Buffer for URLs from the currently parsed sitemap. Used for tracking partially loaded sitemaps across migrations.
|
|
55
|
-
*/
|
|
56
|
-
inProgressEntries: new Set(),
|
|
57
|
-
/**
|
|
58
|
-
* Set of sitemap URLs that have not been parsed yet. If the set is empty and `inProgressSitemapUrl` is `null`, the sitemap loading is finished.
|
|
59
|
-
*/
|
|
60
|
-
pendingSitemapUrls: new Set(),
|
|
61
|
-
};
|
|
62
|
-
/**
|
|
63
|
-
* Object stream of URLs parsed from the sitemaps.
|
|
64
|
-
* Using `highWaterMark`, this can manage the speed of the sitemap loading.
|
|
65
|
-
*
|
|
66
|
-
* Fetch the next URL to be processed using `fetchNextRequest()`.
|
|
67
|
-
*/
|
|
68
|
-
#urlQueueStream;
|
|
69
|
-
/**
|
|
70
|
-
* Indicates whether the request list sitemap loading was aborted.
|
|
71
|
-
*
|
|
72
|
-
* If the loading was aborted before the sitemaps were fully loaded, the request list might be missing some URLs.
|
|
73
|
-
* The `isSitemapFullyLoaded` method can be used to check if the sitemaps were fully loaded.
|
|
74
|
-
*
|
|
75
|
-
* If the loading is aborted and all the requests are handled, `isFinished()` will return `true`.
|
|
76
|
-
*/
|
|
77
|
-
#abortLoading = false;
|
|
78
|
-
/** Number of URLs that were marked as handled */
|
|
79
|
-
#handledUrlCount = 0;
|
|
80
|
-
#persistStateKey;
|
|
81
|
-
#store;
|
|
82
|
-
#closed = false;
|
|
83
|
-
/**
|
|
84
|
-
* Proxy URL to be used for sitemap loading.
|
|
85
|
-
*/
|
|
86
|
-
#proxyUrl;
|
|
87
|
-
/**
|
|
88
|
-
* Enqueue strategy applied to sitemap-derived URLs and stamped onto the emitted `Request` objects.
|
|
89
|
-
*/
|
|
90
|
-
#enqueueStrategy;
|
|
91
|
-
/**
|
|
92
|
-
* Logger instance.
|
|
93
|
-
*/
|
|
94
|
-
#log;
|
|
95
|
-
#urlExcludePatternObjects = [];
|
|
96
|
-
#urlPatternObjects = [];
|
|
97
|
-
/** EventManager used to handle persistence */
|
|
98
|
-
#events;
|
|
99
|
-
#persistenceOptions;
|
|
100
|
-
/** @internal */
|
|
101
|
-
constructor(options) {
|
|
102
|
-
const { include, exclude, persistStateKey, persistenceOptions, proxyUrl, maxBufferSize, sitemapUrls, enqueueStrategy, } = parseArgument(options, sitemapRequestLoaderOptionsSchema, 'SitemapRequestLoaderOptions');
|
|
103
|
-
this.#log = serviceLocator.getLogger().child({ prefix: 'SitemapRequestLoader' });
|
|
104
|
-
if (exclude?.length) {
|
|
105
|
-
this.#urlExcludePatternObjects.push(...constructUrlPatternObjects(exclude));
|
|
106
|
-
}
|
|
107
|
-
if (include?.length) {
|
|
108
|
-
this.#urlPatternObjects.push(...constructUrlPatternObjects(include));
|
|
109
|
-
}
|
|
110
|
-
this.#persistStateKey = persistStateKey;
|
|
111
|
-
this.#persistenceOptions = { enable: true, ...persistenceOptions };
|
|
112
|
-
this.#proxyUrl = proxyUrl;
|
|
113
|
-
this.#enqueueStrategy = enqueueStrategy;
|
|
114
|
-
this.#urlQueueStream = this.createNewStream(maxBufferSize);
|
|
115
|
-
this.#sitemapParsingProgress.pendingSitemapUrls = new Set(sitemapUrls);
|
|
116
|
-
this.#events = serviceLocator.getEventManager();
|
|
117
|
-
this.persistState = this.persistState.bind(this);
|
|
118
|
-
}
|
|
119
|
-
/**
|
|
120
|
-
* Creates a new object stream with the specified highWaterMark.
|
|
121
|
-
* @param highWaterMark High water mark for the stream (the maximum number of objects the stream will buffer).
|
|
122
|
-
* @returns A new object stream.
|
|
123
|
-
*/
|
|
124
|
-
createNewStream(highWaterMark) {
|
|
125
|
-
return new Transform({
|
|
126
|
-
objectMode: true,
|
|
127
|
-
highWaterMark,
|
|
128
|
-
}).pause();
|
|
129
|
-
}
|
|
130
|
-
/**
|
|
131
|
-
* Returns a function that checks whether the provided pattern matches the closure URL.
|
|
132
|
-
* @param url URL to be checked.
|
|
133
|
-
* @returns A matcher function that checks whether the pattern matches the closure URL.
|
|
134
|
-
*/
|
|
135
|
-
matchesUrl(url) {
|
|
136
|
-
return (patternObject) => {
|
|
137
|
-
const { regexp, glob } = patternObject;
|
|
138
|
-
const matchesRegex = (regexp && url.match(regexp)) || false;
|
|
139
|
-
const matchesGlob = (glob && minimatch(url, glob, { nocase: true })) || false;
|
|
140
|
-
return Boolean(matchesRegex || matchesGlob);
|
|
141
|
-
};
|
|
142
|
-
}
|
|
143
|
-
/**
|
|
144
|
-
* Checks whether the URL matches the `include` / `exclude` patterns provided in the `options`.
|
|
145
|
-
* @param url URL to be checked.
|
|
146
|
-
* @returns `true` if the URL matches the patterns, `false` otherwise.
|
|
147
|
-
*/
|
|
148
|
-
isUrlMatchingPatterns(url) {
|
|
149
|
-
return (!this.#urlExcludePatternObjects.some(this.matchesUrl(url)) &&
|
|
150
|
-
(this.#urlPatternObjects.length === 0 || this.#urlPatternObjects.some(this.matchesUrl(url))));
|
|
151
|
-
}
|
|
152
|
-
/**
|
|
153
|
-
* Adds a URL to the queue of parsed URLs.
|
|
154
|
-
*
|
|
155
|
-
* Blocks if the stream is full until it is drained.
|
|
156
|
-
*/
|
|
157
|
-
async pushNextUrl(url) {
|
|
158
|
-
return new Promise((resolve) => {
|
|
159
|
-
if (this.#closed || (url && !this.isUrlMatchingPatterns(url))) {
|
|
160
|
-
resolve();
|
|
161
|
-
return;
|
|
162
|
-
}
|
|
163
|
-
if (!this.#urlQueueStream.push(url)) {
|
|
164
|
-
// This doesn't work with the 'drain' event (it's not emitted for some reason).
|
|
165
|
-
this.#urlQueueStream.once('readdata', () => {
|
|
166
|
-
resolve();
|
|
167
|
-
});
|
|
168
|
-
}
|
|
169
|
-
else {
|
|
170
|
-
resolve();
|
|
171
|
-
}
|
|
172
|
-
});
|
|
173
|
-
}
|
|
174
|
-
/**
|
|
175
|
-
* Reads the next URL from the queue of parsed URLs.
|
|
176
|
-
*
|
|
177
|
-
* If the stream is empty, blocks until a new URL is pushed.
|
|
178
|
-
* @returns The next URL from the queue or `null` if we have read all URLs.
|
|
179
|
-
*/
|
|
180
|
-
async readNextUrl() {
|
|
181
|
-
return new Promise((resolve) => {
|
|
182
|
-
if (this.#closed) {
|
|
183
|
-
resolve(null);
|
|
184
|
-
return;
|
|
185
|
-
}
|
|
186
|
-
const result = this.#urlQueueStream.read();
|
|
187
|
-
if (!result && !this.isSitemapFullyLoaded()) {
|
|
188
|
-
this.#urlQueueStream.once('readable', () => {
|
|
189
|
-
const nextUrl = this.#urlQueueStream.read();
|
|
190
|
-
resolve(nextUrl);
|
|
191
|
-
});
|
|
192
|
-
}
|
|
193
|
-
else {
|
|
194
|
-
resolve(result);
|
|
195
|
-
}
|
|
196
|
-
this.#urlQueueStream.emit('readdata');
|
|
197
|
-
});
|
|
198
|
-
}
|
|
199
|
-
/**
|
|
200
|
-
* Indicates whether the background processing of sitemap contents has successfully finished.
|
|
201
|
-
*
|
|
202
|
-
* If this is `false`, the background processing is either still in progress or was aborted.
|
|
203
|
-
*/
|
|
204
|
-
isSitemapFullyLoaded() {
|
|
205
|
-
return (this.#sitemapParsingProgress.inProgressSitemapUrl === null &&
|
|
206
|
-
this.#sitemapParsingProgress.pendingSitemapUrls.size === 0);
|
|
207
|
-
}
|
|
208
|
-
/**
|
|
209
|
-
* Start processing the sitemaps and loading the URLs.
|
|
210
|
-
*
|
|
211
|
-
* Resolves once all the sitemaps URLs have been fully loaded (sets `isSitemapFullyLoaded` to `true`).
|
|
212
|
-
*/
|
|
213
|
-
async load({ parseSitemapOptions, }) {
|
|
214
|
-
while (!this.isSitemapFullyLoaded() && !this.#abortLoading) {
|
|
215
|
-
const sitemapUrl = this.#sitemapParsingProgress.inProgressSitemapUrl ??
|
|
216
|
-
this.#sitemapParsingProgress.pendingSitemapUrls.values().next().value;
|
|
217
|
-
try {
|
|
218
|
-
for await (const item of parseSitemap([{ type: 'url', url: sitemapUrl }], this.#proxyUrl, {
|
|
219
|
-
...parseSitemapOptions,
|
|
220
|
-
maxDepth: 0,
|
|
221
|
-
emitNestedSitemaps: true,
|
|
222
|
-
enqueueStrategy: this.#enqueueStrategy,
|
|
223
|
-
})) {
|
|
224
|
-
if (!item.originSitemapUrl) {
|
|
225
|
-
// This is a nested sitemap
|
|
226
|
-
this.#sitemapParsingProgress.pendingSitemapUrls.add(item.loc);
|
|
227
|
-
continue;
|
|
228
|
-
}
|
|
229
|
-
if (!this.#sitemapParsingProgress.inProgressEntries.has(item.loc)) {
|
|
230
|
-
await this.pushNextUrl(item.loc);
|
|
231
|
-
this.#sitemapParsingProgress.inProgressEntries.add(item.loc);
|
|
232
|
-
}
|
|
233
|
-
}
|
|
234
|
-
}
|
|
235
|
-
catch (e) {
|
|
236
|
-
this.#log.error('Error loading sitemap contents:', e);
|
|
237
|
-
}
|
|
238
|
-
this.#sitemapParsingProgress.pendingSitemapUrls.delete(sitemapUrl);
|
|
239
|
-
this.#sitemapParsingProgress.inProgressEntries.clear();
|
|
240
|
-
this.#sitemapParsingProgress.inProgressSitemapUrl = null;
|
|
241
|
-
}
|
|
242
|
-
this.#urlQueueStream.end();
|
|
243
|
-
}
|
|
244
|
-
/**
|
|
245
|
-
* Open a sitemap and start processing it.
|
|
246
|
-
*
|
|
247
|
-
* Resolves to a new instance of `SitemapRequestLoader`, which **might not be fully loaded yet** - i.e. the sitemap might still be loading in the background.
|
|
248
|
-
*
|
|
249
|
-
* Track the loading progress using the `isSitemapFullyLoaded` property.
|
|
250
|
-
*/
|
|
251
|
-
static async open(options) {
|
|
252
|
-
const { httpClient, ...restOptions } = options;
|
|
253
|
-
const requestList = new SitemapRequestLoader({
|
|
254
|
-
...restOptions,
|
|
255
|
-
persistStateKey: options.persistStateKey ?? STATE_PERSISTENCE_KEY,
|
|
256
|
-
});
|
|
257
|
-
await requestList.restoreState();
|
|
258
|
-
void requestList.load({
|
|
259
|
-
parseSitemapOptions: { logger: serviceLocator.getLogger(), ...options.parseSitemapOptions, httpClient },
|
|
260
|
-
});
|
|
261
|
-
if (requestList.#persistenceOptions.enable) {
|
|
262
|
-
requestList.#events.on(EventType.PERSIST_STATE, requestList.persistState);
|
|
263
|
-
}
|
|
264
|
-
options?.signal?.addEventListener('abort', () => {
|
|
265
|
-
requestList.#abortLoading = true;
|
|
266
|
-
});
|
|
267
|
-
if (options.timeoutMillis) {
|
|
268
|
-
setTimeout(() => {
|
|
269
|
-
requestList.#abortLoading = true;
|
|
270
|
-
}, options.timeoutMillis);
|
|
271
|
-
}
|
|
272
|
-
return requestList;
|
|
273
|
-
}
|
|
274
|
-
/**
|
|
275
|
-
* @inheritDoc
|
|
276
|
-
*/
|
|
277
|
-
async getTotalCount() {
|
|
278
|
-
// Total known so far = not-yet-fetched (still buffered in the stream) + in-progress (fetched but not
|
|
279
|
-
// yet handled) + already handled.
|
|
280
|
-
return this.#urlQueueStream.readableLength + this.inProgress.size + this.#handledUrlCount;
|
|
281
|
-
}
|
|
282
|
-
/**
|
|
283
|
-
* @inheritDoc
|
|
284
|
-
*/
|
|
285
|
-
async getPendingCount() {
|
|
286
|
-
// Pending = everything not yet handled = not-yet-fetched + in-progress.
|
|
287
|
-
return this.#urlQueueStream.readableLength + this.inProgress.size;
|
|
288
|
-
}
|
|
289
|
-
/**
|
|
290
|
-
* Combines this list with a request manager (a {@link RequestQueue} by default) into a
|
|
291
|
-
* {@link RequestManagerTandem}, allowing requests to be added and reclaimed while still
|
|
292
|
-
* being read from this list first.
|
|
293
|
-
*/
|
|
294
|
-
async toTandem(requestManager) {
|
|
295
|
-
// Import here to avoid circular imports.
|
|
296
|
-
const { RequestManagerTandem } = await import('./request_manager_tandem.js');
|
|
297
|
-
const { RequestQueue } = await import('./request_queue.js');
|
|
298
|
-
return new RequestManagerTandem(this, requestManager ?? (await RequestQueue.open()));
|
|
299
|
-
}
|
|
300
|
-
/**
|
|
301
|
-
* @inheritDoc
|
|
302
|
-
*/
|
|
303
|
-
async isFinished() {
|
|
304
|
-
return ((await this.isEmpty()) && this.inProgress.size === 0 && (this.isSitemapFullyLoaded() || this.#abortLoading));
|
|
305
|
-
}
|
|
306
|
-
/**
|
|
307
|
-
* @inheritDoc
|
|
308
|
-
*/
|
|
309
|
-
async isEmpty() {
|
|
310
|
-
return this.#urlQueueStream.readableLength === 0;
|
|
311
|
-
}
|
|
312
|
-
/**
|
|
313
|
-
* @inheritDoc
|
|
314
|
-
*/
|
|
315
|
-
async getHandledCount() {
|
|
316
|
-
return this.#handledUrlCount;
|
|
317
|
-
}
|
|
318
|
-
/**
|
|
319
|
-
* @inheritDoc
|
|
320
|
-
*/
|
|
321
|
-
async persistState() {
|
|
322
|
-
if (this.#persistStateKey === undefined) {
|
|
323
|
-
return;
|
|
324
|
-
}
|
|
325
|
-
this.#store ??= await KeyValueStore.open();
|
|
326
|
-
const urlQueue = [];
|
|
327
|
-
while (this.#urlQueueStream.readableLength > 0) {
|
|
328
|
-
const url = this.#urlQueueStream.read();
|
|
329
|
-
if (url === null) {
|
|
330
|
-
break;
|
|
331
|
-
}
|
|
332
|
-
urlQueue.push(url);
|
|
333
|
-
}
|
|
334
|
-
// Create a new stream, as we have read all the URLs from the current one.
|
|
335
|
-
// Pushing the urls back to the original stream might not be possible if it has been ended.
|
|
336
|
-
const previousStream = this.#urlQueueStream;
|
|
337
|
-
const newStream = this.createNewStream(previousStream.readableHighWaterMark);
|
|
338
|
-
for (const url of urlQueue) {
|
|
339
|
-
newStream.push(url);
|
|
340
|
-
}
|
|
341
|
-
if (previousStream.writableEnded) {
|
|
342
|
-
newStream.end();
|
|
343
|
-
}
|
|
344
|
-
this.#urlQueueStream = newStream;
|
|
345
|
-
// A `pushNextUrl()` call may be blocked on backpressure, waiting for a `readdata` event on the
|
|
346
|
-
// previous stream. That event is only ever emitted by `readNextUrl()` on the current stream, so
|
|
347
|
-
// after the swap the waiter would never be notified and the background sitemap loading would hang.
|
|
348
|
-
// Re-emit `readdata` on the previous stream to release any such pending waiter (its URL has already
|
|
349
|
-
// been transferred to the new stream above).
|
|
350
|
-
previousStream.emit('readdata');
|
|
351
|
-
await this.#store.setValue(this.#persistStateKey, {
|
|
352
|
-
sitemapParsingProgress: {
|
|
353
|
-
pendingSitemapUrls: Array.from(this.#sitemapParsingProgress.pendingSitemapUrls),
|
|
354
|
-
inProgressSitemapUrl: this.#sitemapParsingProgress.inProgressSitemapUrl,
|
|
355
|
-
inProgressEntries: Array.from(this.#sitemapParsingProgress.inProgressEntries),
|
|
356
|
-
},
|
|
357
|
-
// Re-queue in-progress requests to the front so they are retried if the state is restored.
|
|
358
|
-
urlQueue: [...this.inProgress, ...urlQueue],
|
|
359
|
-
requestData: Array.from(this.#requestData.entries()),
|
|
360
|
-
abortLoading: this.#abortLoading,
|
|
361
|
-
closed: this.#closed,
|
|
362
|
-
});
|
|
363
|
-
}
|
|
364
|
-
async restoreState() {
|
|
365
|
-
await purgeDefaultStorages({ onlyPurgeOnce: true });
|
|
366
|
-
if (this.#persistStateKey === undefined) {
|
|
367
|
-
return;
|
|
368
|
-
}
|
|
369
|
-
this.#store ??= await KeyValueStore.open();
|
|
370
|
-
const state = await this.#store.getValue(this.#persistStateKey);
|
|
371
|
-
if (state === null) {
|
|
372
|
-
return;
|
|
373
|
-
}
|
|
374
|
-
this.#sitemapParsingProgress = {
|
|
375
|
-
pendingSitemapUrls: new Set(state.sitemapParsingProgress.pendingSitemapUrls),
|
|
376
|
-
inProgressSitemapUrl: state.sitemapParsingProgress.inProgressSitemapUrl,
|
|
377
|
-
inProgressEntries: new Set(state.sitemapParsingProgress.inProgressEntries),
|
|
378
|
-
};
|
|
379
|
-
this.#requestData = new Map(state.requestData ?? []);
|
|
380
|
-
for (const url of state.urlQueue) {
|
|
381
|
-
this.#urlQueueStream.push(url);
|
|
382
|
-
}
|
|
383
|
-
this.#abortLoading = state.abortLoading;
|
|
384
|
-
this.#closed = state.closed;
|
|
385
|
-
}
|
|
386
|
-
/**
|
|
387
|
-
* @inheritDoc
|
|
388
|
-
*/
|
|
389
|
-
async fetchNextRequest() {
|
|
390
|
-
const nextUrl = await this.readNextUrl();
|
|
391
|
-
if (!nextUrl) {
|
|
392
|
-
return null;
|
|
393
|
-
}
|
|
394
|
-
// A restored in-progress request already has its Request data; don't overwrite it.
|
|
395
|
-
if (!this.#requestData.has(nextUrl)) {
|
|
396
|
-
this.#requestData.set(nextUrl, new Request({ url: nextUrl, enqueueStrategy: this.#enqueueStrategy }));
|
|
397
|
-
}
|
|
398
|
-
this.inProgress.add(nextUrl);
|
|
399
|
-
return this.#requestData.get(nextUrl);
|
|
400
|
-
}
|
|
401
|
-
/**
|
|
402
|
-
* @inheritDoc
|
|
403
|
-
*/
|
|
404
|
-
async *[Symbol.asyncIterator]() {
|
|
405
|
-
while (!(await this.isFinished())) {
|
|
406
|
-
const request = await this.fetchNextRequest();
|
|
407
|
-
if (!request)
|
|
408
|
-
break;
|
|
409
|
-
yield request;
|
|
410
|
-
}
|
|
411
|
-
}
|
|
412
|
-
/**
|
|
413
|
-
* Aborts the internal sitemap loading, stops the processing of the sitemap contents and drops all the pending URLs.
|
|
414
|
-
*
|
|
415
|
-
* Calling `fetchNextRequest()` after this method will always return `null`.
|
|
416
|
-
*/
|
|
417
|
-
async teardown() {
|
|
418
|
-
this.#closed = true;
|
|
419
|
-
this.#abortLoading = true;
|
|
420
|
-
this.#events.off(EventType.PERSIST_STATE, this.persistState);
|
|
421
|
-
await this.persistState();
|
|
422
|
-
this.#urlQueueStream.emit('readdata'); // unblocks the potentially waiting `pushNextUrl` call
|
|
423
|
-
}
|
|
424
|
-
/**
|
|
425
|
-
* @inheritDoc
|
|
426
|
-
*/
|
|
427
|
-
async markRequestAsHandled(request) {
|
|
428
|
-
this.#handledUrlCount += 1;
|
|
429
|
-
this.ensureInProgress(request.url);
|
|
430
|
-
this.inProgress.delete(request.url);
|
|
431
|
-
this.#requestData.delete(request.url);
|
|
432
|
-
}
|
|
433
|
-
ensureInProgress(url) {
|
|
434
|
-
if (!this.inProgress.has(url)) {
|
|
435
|
-
throw new Error(`The request is not being processed (url: ${url})`);
|
|
436
|
-
}
|
|
437
|
-
}
|
|
438
|
-
}
|
|
@@ -1,239 +0,0 @@
|
|
|
1
|
-
import type { Dictionary } from '@crawlee/types';
|
|
2
|
-
import type { Configuration } from '../configuration.js';
|
|
3
|
-
import type { Request, Source } from '../request.js';
|
|
4
|
-
import type { IRequestManager, RequestsLike } from './request_manager.js';
|
|
5
|
-
import type { AddRequestsBatchedOptions, AddRequestsBatchedResult, RequestQueueOperationInfo, RequestQueueOperationOptions } from './request_queue.js';
|
|
6
|
-
import type { StorageIdentifier } from './storage_instance_manager.js';
|
|
7
|
-
import type { StorageOpenOptions } from './utils.js';
|
|
8
|
-
/**
|
|
9
|
-
* Opens a request manager, matching the shape of storage `open` methods such as
|
|
10
|
-
* {@link RequestQueue.open|`RequestQueue.open`}.
|
|
11
|
-
*
|
|
12
|
-
* {@link ThrottlingRequestManager} calls this once per configured domain, so every per-domain queue shares the
|
|
13
|
-
* concrete type and storage backend of the manager being wrapped.
|
|
14
|
-
*/
|
|
15
|
-
export type RequestManagerOpener<T extends IRequestManager = IRequestManager> = (identifier: string | StorageIdentifier, options?: StorageOpenOptions) => Promise<T>;
|
|
16
|
-
/**
|
|
17
|
-
* A request manager that can pace requests per domain, as {@link ThrottlingRequestManager} does.
|
|
18
|
-
*
|
|
19
|
-
* The crawlers detect this structurally rather than by type, so a wrapper can opt in by forwarding these three
|
|
20
|
-
* methods without {@link IRequestManager} having to know that throttling exists.
|
|
21
|
-
*/
|
|
22
|
-
export interface SupportsDomainThrottling {
|
|
23
|
-
/** @see {@link ThrottlingRequestManager.recordDomainDelay} */
|
|
24
|
-
recordDomainDelay(url: string, retryAfterMs?: number | null): boolean;
|
|
25
|
-
/** @see {@link ThrottlingRequestManager.setCrawlDelay} */
|
|
26
|
-
setCrawlDelay(url: string, delaySeconds: number): boolean;
|
|
27
|
-
/** @see {@link ThrottlingRequestManager.assertNoStalledDomains} */
|
|
28
|
-
assertNoStalledDomains(): Promise<void>;
|
|
29
|
-
}
|
|
30
|
-
/** Whether `manager` can pace requests per domain. */
|
|
31
|
-
export declare function supportsDomainThrottling(manager: unknown): manager is SupportsDomainThrottling;
|
|
32
|
-
/** Options for {@link ThrottlingRequestManager}. */
|
|
33
|
-
export interface ThrottlingRequestManagerOptions<T extends IRequestManager = IRequestManager> {
|
|
34
|
-
/**
|
|
35
|
-
* The request manager to wrap, usually a {@link RequestQueue}. Requests for domains that are not throttled
|
|
36
|
-
* are stored here.
|
|
37
|
-
*/
|
|
38
|
-
inner: T;
|
|
39
|
-
/**
|
|
40
|
-
* Which domains to throttle: a list of hostnames, or `'all'` for every domain the crawl encounters.
|
|
41
|
-
*
|
|
42
|
-
* Matching a listed hostname is case-insensitive and exact - wildcards such as `*.example.com` are not
|
|
43
|
-
* supported, so list each subdomain you care about (or set
|
|
44
|
-
* {@link ThrottlingRequestManagerOptions.throttleBy|`throttleBy: 'registrableDomain'`}). An
|
|
45
|
-
* internationalized domain may be given in either its unicode or its punycode form, and an IPv6 address has
|
|
46
|
-
* to be bracketed (`[::1]`). Requests for any other domain bypass throttling entirely.
|
|
47
|
-
*
|
|
48
|
-
* `'all'` gives each domain a queue of its own the first time it is seen, so that it can be held back
|
|
49
|
-
* without its requests being repeatedly popped and re-enqueued. One request queue per domain is not free,
|
|
50
|
-
* which is what {@link ThrottlingRequestManagerOptions.maxThrottledDomains|`maxThrottledDomains`} is
|
|
51
|
-
* there to bound.
|
|
52
|
-
*/
|
|
53
|
-
domains: string[] | 'all';
|
|
54
|
-
/**
|
|
55
|
-
* A floor under the crawl delay of every throttled domain, in seconds - the proactive clock described on
|
|
56
|
-
* {@link ThrottlingRequestManager}. A domain whose robots.txt asks for a longer `Crawl-delay` gets the
|
|
57
|
-
* longer one; this is a minimum, not an override.
|
|
58
|
-
* @default 0
|
|
59
|
-
*/
|
|
60
|
-
minCrawlDelaySecs?: number;
|
|
61
|
-
/**
|
|
62
|
-
* What counts as "the same domain": the exact hostname, or the registrable domain it belongs to
|
|
63
|
-
* (`example.com` for `www.example.com`, `a.example.co.uk` and so on). Hosts with no registrable domain -
|
|
64
|
-
* IP addresses, `localhost` - are always throttled per hostname.
|
|
65
|
-
*
|
|
66
|
-
* Grouping by registrable domain gives subdomains a single pair of clocks and a single queue, which is what
|
|
67
|
-
* you want when the pacing is there to be polite to one server rather than to satisfy a specific host's
|
|
68
|
-
* rate limit.
|
|
69
|
-
* @default 'hostname'
|
|
70
|
-
*/
|
|
71
|
-
throttleBy?: 'hostname' | 'registrableDomain';
|
|
72
|
-
/**
|
|
73
|
-
* The most domains a run may throttle at once. Exceeding it throws, rather than silently letting the
|
|
74
|
-
* throttling lapse - one request queue per domain is not free, and a crawl that discovers domains without
|
|
75
|
-
* bound would drown the storage backend in them.
|
|
76
|
-
*
|
|
77
|
-
* Only domains discovered under `domains: 'all'` count against this; an explicit list is taken at face value.
|
|
78
|
-
* @default 100
|
|
79
|
-
*/
|
|
80
|
-
maxThrottledDomains?: number;
|
|
81
|
-
/**
|
|
82
|
-
* The key under which the discovered domain list is kept in the default key-value store, so that a restart
|
|
83
|
-
* with `purgeOnStart` disabled reopens their queues instead of stranding whatever they still hold. Only
|
|
84
|
-
* written under `domains: 'all'`.
|
|
85
|
-
*
|
|
86
|
-
* Give each manager its own key when running several of them against the same storage.
|
|
87
|
-
* @default 'CRAWLEE_THROTTLED_DOMAINS'
|
|
88
|
-
*/
|
|
89
|
-
persistStateKey?: string;
|
|
90
|
-
/**
|
|
91
|
-
* Opens the per-domain queues, one per throttled domain, each under the alias `throttled-<domain>`.
|
|
92
|
-
* @default RequestQueue.open
|
|
93
|
-
*/
|
|
94
|
-
requestManagerOpener?: RequestManagerOpener<T>;
|
|
95
|
-
/**
|
|
96
|
-
* The delay applied after a domain's first HTTP 429, doubled on each subsequent one.
|
|
97
|
-
* @default 2
|
|
98
|
-
*/
|
|
99
|
-
baseDelaySecs?: number;
|
|
100
|
-
/**
|
|
101
|
-
* Upper bound on the delay between requests to a rate-limited domain, applied to both the exponential
|
|
102
|
-
* backoff and a `Retry-After` value.
|
|
103
|
-
* @default 60
|
|
104
|
-
*/
|
|
105
|
-
maxDelaySecs?: number;
|
|
106
|
-
/**
|
|
107
|
-
* How long a domain may rate-limit us without a single request getting through before the crawl is
|
|
108
|
-
* abandoned with a {@link PersistentRateLimitError}.
|
|
109
|
-
*
|
|
110
|
-
* A domain that keeps answering 429 for this long is not going to be crawled by waiting longer - the
|
|
111
|
-
* concurrency is too high for it, or it has blocked us outright. Its requests are deliberately left in
|
|
112
|
-
* their queue, so re-running the crawl with `purgeOnStart` disabled picks them up once the domain recovers.
|
|
113
|
-
*
|
|
114
|
-
* A crawler running with `keepAlive` is exempt - outliving a domain that will not let us through is the
|
|
115
|
-
* whole point there.
|
|
116
|
-
* @default 900
|
|
117
|
-
*/
|
|
118
|
-
maxDomainStallSecs?: number;
|
|
119
|
-
}
|
|
120
|
-
/**
|
|
121
|
-
* A request manager that wraps another one and paces requests per domain.
|
|
122
|
-
*
|
|
123
|
-
* Requests for a throttled domain are routed into their own queue when they are added, so each request lives in
|
|
124
|
-
* exactly one place and deduplication keeps working. Everything else goes to the wrapped manager untouched.
|
|
125
|
-
*
|
|
126
|
-
* {@link ThrottlingRequestManager.fetchNextRequest|`fetchNextRequest()`} serves the domain that has been waiting
|
|
127
|
-
* longest and skips any that are backing off, falling back to the wrapped manager. It never blocks: while every
|
|
128
|
-
* remaining request belongs to a throttled domain it returns `null` and {@link ThrottlingRequestManager.isEmpty}
|
|
129
|
-
* reports `true`, so the crawler idles instead of holding a concurrency slot open.
|
|
130
|
-
*
|
|
131
|
-
* Each throttled domain runs two independent clocks, and may be dispatched to once **both** have run out:
|
|
132
|
-
* - **Backoff**, set by HTTP 429 responses - honouring `Retry-After`, and otherwise doubling from `baseDelaySecs`.
|
|
133
|
-
* Reactive and temporary: it decays once the domain stops turning us away. The crawlers report the 429s
|
|
134
|
-
* themselves; a request held back this way is retried later without counting against `maxRequestRetries` and
|
|
135
|
-
* without penalising its session.
|
|
136
|
-
* - **Crawl delay**, the minimum interval between two dispatches to the domain, armed after each one. Proactive
|
|
137
|
-
* and constant: whatever the domain's robots.txt asks for, floored by
|
|
138
|
-
* {@link ThrottlingRequestManagerOptions.minCrawlDelaySecs|`minCrawlDelaySecs`}. Either may be absent, in
|
|
139
|
-
* which case the other one is the delay.
|
|
140
|
-
*
|
|
141
|
-
* Which domains get those clocks is {@link ThrottlingRequestManagerOptions.domains|`domains`} - a list, or
|
|
142
|
-
* `'all'` for every domain the crawl encounters.
|
|
143
|
-
*
|
|
144
|
-
* **Example usage:**
|
|
145
|
-
*
|
|
146
|
-
* ```ts
|
|
147
|
-
* const crawler = new CheerioCrawler({
|
|
148
|
-
* requestManager: new ThrottlingRequestManager({
|
|
149
|
-
* inner: await RequestQueue.open(),
|
|
150
|
-
* domains: ['api.example.com', 'slow-site.org'],
|
|
151
|
-
* }),
|
|
152
|
-
* requestHandler: async ({ request }) => { ... },
|
|
153
|
-
* });
|
|
154
|
-
* ```
|
|
155
|
-
*
|
|
156
|
-
* @category Sources
|
|
157
|
-
*/
|
|
158
|
-
export declare class ThrottlingRequestManager<T extends IRequestManager = IRequestManager> implements IRequestManager, SupportsDomainThrottling {
|
|
159
|
-
#private;
|
|
160
|
-
private readonly config;
|
|
161
|
-
private readonly domainStates;
|
|
162
|
-
private readonly log;
|
|
163
|
-
constructor(options: ThrottlingRequestManagerOptions<T>, config?: Configuration);
|
|
164
|
-
/** The wrapped manager, holding every request whose domain is not throttled. */
|
|
165
|
-
get innerManager(): T;
|
|
166
|
-
/**
|
|
167
|
-
* Records a 429 response and puts the URL's domain into backoff.
|
|
168
|
-
*
|
|
169
|
-
* @returns `false` if the domain is not configured for throttling, in which case this is a no-op.
|
|
170
|
-
*/
|
|
171
|
-
recordDomainDelay(url: string, retryAfterMs?: number | null): boolean;
|
|
172
|
-
/**
|
|
173
|
-
* Records the `Crawl-delay` a domain's robots.txt asked for, which becomes its crawl delay unless
|
|
174
|
-
* {@link ThrottlingRequestManagerOptions.minCrawlDelaySecs|`minCrawlDelaySecs`} asks for longer.
|
|
175
|
-
*
|
|
176
|
-
* The first value wins, so a robots.txt re-fetch cannot change the cadence mid-crawl.
|
|
177
|
-
*
|
|
178
|
-
* @returns `false` if the domain is not throttled, in which case this is a no-op.
|
|
179
|
-
*/
|
|
180
|
-
setCrawlDelay(url: string, delaySeconds: number): boolean;
|
|
181
|
-
/**
|
|
182
|
-
* Throws {@link PersistentRateLimitError} if any domain has been rate-limiting us past
|
|
183
|
-
* {@link ThrottlingRequestManagerOptions.maxDomainStallSecs|`maxDomainStallSecs`} without letting a single
|
|
184
|
-
* request through.
|
|
185
|
-
*
|
|
186
|
-
* A domain qualifies only while it still has queued requests and is actively rate-limiting - a domain that
|
|
187
|
-
* has simply run out of work is finished, not stalled, and one being waited out under a long robots.txt
|
|
188
|
-
* `Crawl-delay` is being obeyed, not stonewalled.
|
|
189
|
-
*/
|
|
190
|
-
assertNoStalledDomains(): Promise<void>;
|
|
191
|
-
addRequest(requestLike: Source, options?: RequestQueueOperationOptions): Promise<RequestQueueOperationInfo>;
|
|
192
|
-
/**
|
|
193
|
-
* Adds requests in batches, routing each one to the manager that owns its domain.
|
|
194
|
-
*
|
|
195
|
-
* Batching, validation, deduplication and `Retry-After`-free bookkeeping are all delegated to the target
|
|
196
|
-
* managers - this only decides where each request goes, one batch at a time, so a lazy or unbounded input
|
|
197
|
-
* iterable is never fully materialized.
|
|
198
|
-
*/
|
|
199
|
-
addRequestsBatched(requests: RequestsLike, options?: AddRequestsBatchedOptions): Promise<AddRequestsBatchedResult>;
|
|
200
|
-
reclaimRequest(request: Request, options?: RequestQueueOperationOptions): Promise<RequestQueueOperationInfo | null>;
|
|
201
|
-
markRequestAsHandled(request: Request): Promise<RequestQueueOperationInfo | void | null>;
|
|
202
|
-
getTotalCount(): Promise<number>;
|
|
203
|
-
getPendingCount(): Promise<number>;
|
|
204
|
-
getHandledCount(): Promise<number>;
|
|
205
|
-
/**
|
|
206
|
-
* Whether the next {@link ThrottlingRequestManager.fetchNextRequest} would return `null`.
|
|
207
|
-
*
|
|
208
|
-
* Requests waiting on a throttled domain count as unavailable, so a crawler whose task loop is gated on
|
|
209
|
-
* this idles for the backoff instead of spinning on a fetch that cannot succeed yet.
|
|
210
|
-
*/
|
|
211
|
-
isEmpty(): Promise<boolean>;
|
|
212
|
-
/** Unlike {@link ThrottlingRequestManager.isEmpty}, throttled requests still count as outstanding work. */
|
|
213
|
-
isFinished(): Promise<boolean>;
|
|
214
|
-
/**
|
|
215
|
-
* Empties every manager and clears the accumulated backoff. A robots.txt `Crawl-delay` is a property of the
|
|
216
|
-
* site rather than of the run, so it survives.
|
|
217
|
-
*/
|
|
218
|
-
purge(): Promise<void>;
|
|
219
|
-
/**
|
|
220
|
-
* Empties the per-domain queues, leaving the wrapped manager alone.
|
|
221
|
-
*
|
|
222
|
-
* Those queues are this manager's own no matter who owns the one it wraps, which is what makes this safe to
|
|
223
|
-
* call where a full {@link ThrottlingRequestManager.purge|`purge()`} would not be.
|
|
224
|
-
*/
|
|
225
|
-
purgeDomainQueues(): Promise<void>;
|
|
226
|
-
setExpectedRequestProcessingTimeSecs(secs: number): Promise<void>;
|
|
227
|
-
/**
|
|
228
|
-
* Returns the next request from a domain that is not backing off, or from the inner manager.
|
|
229
|
-
*
|
|
230
|
-
* Returns `null` while every remaining request belongs to a throttled domain - it never waits the backoff
|
|
231
|
-
* out, because a consumer parked in here holds a concurrency slot, which the autoscaler reads as spare
|
|
232
|
-
* capacity and answers by scaling up. Callers poll instead, and {@link ThrottlingRequestManager.isEmpty}
|
|
233
|
-
* reports `true` meanwhile so the crawler's task loop idles rather than spins.
|
|
234
|
-
*/
|
|
235
|
-
fetchNextRequest<R extends Dictionary = Dictionary>(): Promise<Request<R> | null>;
|
|
236
|
-
[Symbol.asyncIterator](): AsyncGenerator<Request<Dictionary>, void, unknown>;
|
|
237
|
-
persistState(): Promise<void>;
|
|
238
|
-
drop(): Promise<void>;
|
|
239
|
-
}
|