@crawlee/core 4.0.0-beta.99 → 4.0.0-rc.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/configuration.d.ts +16 -47
- package/configuration.js +13 -25
- package/debug.js +4 -4
- package/errors.d.ts +28 -38
- package/errors.js +33 -47
- package/events/event_manager.d.ts +2 -2
- package/events/event_manager.js +7 -6
- package/events/index.d.ts +1 -0
- package/events/local_event_manager.d.ts +1 -8
- package/events/local_event_manager.js +13 -13
- package/events/system_info.d.ts +38 -0
- package/index.d.ts +2 -8
- package/index.js +4 -8
- package/internal.d.ts +8 -0
- package/internal.js +9 -0
- package/log.d.ts +10 -11
- package/log.js +52 -20
- package/memory-storage/memory-storage.d.ts +15 -18
- package/memory-storage/memory-storage.js +80 -58
- package/memory-storage/resource-clients/dataset.d.ts +1 -6
- package/memory-storage/resource-clients/dataset.js +23 -31
- package/memory-storage/resource-clients/key-value-store.d.ts +1 -10
- package/memory-storage/resource-clients/key-value-store.js +43 -67
- package/memory-storage/resource-clients/request-queue.d.ts +1 -42
- package/memory-storage/resource-clients/request-queue.js +109 -117
- package/owned_or_injected.d.ts +1 -3
- package/owned_or_injected.js +17 -17
- package/package.json +17 -20
- package/proxy_configuration.d.ts +21 -26
- package/proxy_configuration.js +35 -25
- package/recoverable_state.d.ts +104 -47
- package/recoverable_state.js +199 -74
- package/request.d.ts +20 -107
- package/request.js +78 -244
- package/serialization.js +17 -16
- package/service_locator.d.ts +22 -10
- package/service_locator.js +59 -48
- package/storages/batched_adds.d.ts +37 -0
- package/storages/batched_adds.js +73 -0
- package/storages/dataset.d.ts +13 -8
- package/storages/dataset.js +149 -40
- package/storages/index.d.ts +4 -4
- package/storages/index.js +2 -4
- package/storages/key_value_store.d.ts +16 -35
- package/storages/key_value_store.js +223 -110
- package/storages/key_value_store_codec.js +6 -11
- package/storages/request_dedup_cache.d.ts +1 -4
- package/storages/request_dedup_cache.js +15 -15
- package/storages/request_list.d.ts +9 -104
- package/storages/request_list.js +236 -233
- package/storages/request_loader.d.ts +49 -18
- package/storages/request_loader.js +36 -1
- package/storages/request_manager.d.ts +86 -0
- package/storages/request_manager_tandem.d.ts +14 -38
- package/storages/request_manager_tandem.js +67 -64
- package/storages/request_queue.d.ts +23 -50
- package/storages/request_queue.js +371 -226
- package/storages/storage_instance_manager.d.ts +2 -4
- package/storages/storage_instance_manager.js +21 -21
- package/storages/storage_stats.d.ts +1 -1
- package/storages/storage_stats.js +4 -4
- package/storages/transaction.d.ts +270 -0
- package/storages/transaction.js +296 -0
- package/storages/utils.d.ts +6 -3
- package/storages/utils.js +11 -2
- package/system-info/runtime.js +7 -7
- package/url.d.ts +9 -0
- package/url.js +11 -0
- package/validators.d.ts +23 -25
- package/validators.js +14 -25
- package/autoscaling/autoscaled_pool.d.ts +0 -213
- package/autoscaling/autoscaled_pool.js +0 -378
- package/autoscaling/client_load_signal.d.ts +0 -59
- package/autoscaling/client_load_signal.js +0 -73
- package/autoscaling/concurrency_system.d.ts +0 -283
- package/autoscaling/concurrency_system.js +0 -350
- package/autoscaling/cpu_load_signal.d.ts +0 -44
- package/autoscaling/cpu_load_signal.js +0 -46
- package/autoscaling/event_loop_load_signal.d.ts +0 -54
- package/autoscaling/event_loop_load_signal.js +0 -60
- package/autoscaling/index.d.ts +0 -9
- package/autoscaling/index.js +0 -9
- package/autoscaling/load_signal.d.ts +0 -99
- package/autoscaling/load_signal.js +0 -103
- package/autoscaling/memory_load_signal.d.ts +0 -56
- package/autoscaling/memory_load_signal.js +0 -106
- package/autoscaling/snapshotter.d.ts +0 -87
- package/autoscaling/snapshotter.js +0 -67
- package/autoscaling/system_status.d.ts +0 -161
- package/autoscaling/system_status.js +0 -139
- package/autoscaling/weighted_avg.d.ts +0 -5
- package/autoscaling/weighted_avg.js +0 -14
- package/cookie_utils.d.ts +0 -44
- package/cookie_utils.js +0 -122
- package/crawlers/context_pipeline.d.ts +0 -70
- package/crawlers/context_pipeline.js +0 -122
- package/crawlers/crawler_commons.d.ts +0 -257
- package/crawlers/crawler_commons.js +0 -107
- package/crawlers/error_snapshotter.d.ts +0 -59
- package/crawlers/error_snapshotter.js +0 -117
- package/crawlers/error_tracker.d.ts +0 -54
- package/crawlers/error_tracker.js +0 -308
- package/crawlers/index.d.ts +0 -5
- package/crawlers/index.js +0 -5
- package/crawlers/internals/types.d.ts +0 -7
- package/crawlers/statistics.d.ts +0 -209
- package/crawlers/statistics.js +0 -350
- package/enqueue_links/enqueue_links.d.ts +0 -264
- package/enqueue_links/enqueue_links.js +0 -271
- package/enqueue_links/index.d.ts +0 -2
- package/enqueue_links/index.js +0 -2
- package/enqueue_links/shared.d.ts +0 -83
- package/enqueue_links/shared.js +0 -221
- package/router.d.ts +0 -309
- package/router.js +0 -309
- package/session_pool/consts.d.ts +0 -3
- package/session_pool/consts.js +0 -3
- package/session_pool/errors.d.ts +0 -7
- package/session_pool/errors.js +0 -11
- package/session_pool/fingerprint.d.ts +0 -9
- package/session_pool/fingerprint.js +0 -30
- package/session_pool/index.d.ts +0 -4
- package/session_pool/index.js +0 -4
- package/session_pool/session.d.ts +0 -161
- package/session_pool/session.js +0 -218
- package/session_pool/session_pool.d.ts +0 -246
- package/session_pool/session_pool.js +0 -386
- package/storages/access_checking.d.ts +0 -12
- package/storages/access_checking.js +0 -17
- package/storages/sitemap_request_loader.d.ts +0 -249
- package/storages/sitemap_request_loader.js +0 -432
- /package/{crawlers/internals/types.js → events/system_info.js} +0 -0
|
@@ -1,13 +1,17 @@
|
|
|
1
1
|
import { inspect } from 'node:util';
|
|
2
|
-
import {
|
|
3
|
-
import
|
|
2
|
+
import { isAsyncIterable, isIterable } from '@crawlee/utils/internal';
|
|
3
|
+
import { downloadListOfUrls } from '@crawlee/utils';
|
|
4
|
+
import { z } from 'zod';
|
|
4
5
|
import { LruCache } from '@apify/datastructures';
|
|
6
|
+
import { tryCancel } from '@apify/timeout';
|
|
5
7
|
import { Configuration } from '../configuration.js';
|
|
6
8
|
import { getObjectType } from '../debug.js';
|
|
7
|
-
import {
|
|
9
|
+
import { EventType } from '../events/event_manager.js';
|
|
8
10
|
import { Request } from '../request.js';
|
|
9
11
|
import { serviceLocator } from '../service_locator.js';
|
|
10
|
-
import {
|
|
12
|
+
import { parseArgument, schemas, validators } from '../validators.js';
|
|
13
|
+
import { activeStorageTransaction, rejectOperationInTransaction } from './transaction.js';
|
|
14
|
+
import { drainRequestBatches } from './batched_adds.js';
|
|
11
15
|
import { StorageStatsTracker } from './storage_stats.js';
|
|
12
16
|
import { resolveStorageIdentifier } from './storage_instance_manager.js';
|
|
13
17
|
import { getRequestId, purgeDefaultStorages } from './utils.js';
|
|
@@ -17,6 +21,44 @@ import { RequestDeduplicationCache } from './request_dedup_cache.js';
|
|
|
17
21
|
* @internal
|
|
18
22
|
*/
|
|
19
23
|
const MAX_CACHED_REQUESTS = 2_000_000;
|
|
24
|
+
const iterableSchema = z.custom((value) => isIterable(value) || isAsyncIterable(value), {
|
|
25
|
+
error: (issue) => `Expected an iterable or async iterable, got ${getObjectType(issue.input)}`,
|
|
26
|
+
});
|
|
27
|
+
const operationOptionsSchema = z.strictObject({
|
|
28
|
+
forefront: z.boolean().default(false),
|
|
29
|
+
});
|
|
30
|
+
const addRequestsOptionsSchema = z.strictObject({
|
|
31
|
+
forefront: z.boolean().default(false),
|
|
32
|
+
cache: z.boolean().default(true),
|
|
33
|
+
});
|
|
34
|
+
const addRequestsBatchedOptionsSchema = z.strictObject({
|
|
35
|
+
forefront: z.boolean().optional(),
|
|
36
|
+
waitForAllRequestsToBeAdded: z.boolean().default(false),
|
|
37
|
+
batchSize: schemas.anyNumber.default(1000),
|
|
38
|
+
waitBetweenBatchesMillis: schemas.anyNumber.default(1000),
|
|
39
|
+
maxNewRequests: schemas.anyNumber.optional(),
|
|
40
|
+
});
|
|
41
|
+
// Compiled: these run once per request.
|
|
42
|
+
const newRequestLikeSchema = z.compile(z.looseObject({
|
|
43
|
+
url: z.string(),
|
|
44
|
+
id: z.undefined().optional(),
|
|
45
|
+
}));
|
|
46
|
+
const handledRequestSchema = z.compile(z.looseObject({
|
|
47
|
+
id: z.string(),
|
|
48
|
+
uniqueKey: z.string(),
|
|
49
|
+
handledAt: z.string().optional(),
|
|
50
|
+
}));
|
|
51
|
+
const reclaimedRequestSchema = z.compile(z.looseObject({
|
|
52
|
+
id: z.string(),
|
|
53
|
+
uniqueKey: z.string(),
|
|
54
|
+
}));
|
|
55
|
+
const uniqueKeySchema = z.string();
|
|
56
|
+
const openOptionsSchema = z.strictObject({
|
|
57
|
+
configuration: z.instanceof(Configuration).optional(),
|
|
58
|
+
storageBackend: validators.storageBackend.optional(),
|
|
59
|
+
proxyConfiguration: validators.proxyConfiguration.optional(),
|
|
60
|
+
httpClient: schemas.httpClient.optional(),
|
|
61
|
+
});
|
|
20
62
|
/**
|
|
21
63
|
* Represents a queue of URLs to crawl, which is used for deep crawling of websites
|
|
22
64
|
* where you start with several URLs and then recursively
|
|
@@ -55,26 +97,26 @@ export class RequestQueue {
|
|
|
55
97
|
id;
|
|
56
98
|
name;
|
|
57
99
|
backend;
|
|
58
|
-
proxyConfiguration;
|
|
100
|
+
#proxyConfiguration;
|
|
59
101
|
log;
|
|
60
|
-
requestCache;
|
|
102
|
+
#requestCache;
|
|
61
103
|
/**
|
|
62
104
|
* Remembers the `requestId` of every request already submitted to the client — including background
|
|
63
105
|
* batches that `requestCache` skips — so overlapping URL sets aren't re-submitted.
|
|
64
106
|
* See {@link RequestDeduplicationCache} for why this is a separate, cheaper cache.
|
|
65
107
|
*/
|
|
66
|
-
requestSeenCache;
|
|
67
|
-
queuePausedForMigration = false;
|
|
68
|
-
inProgressRequestBatchCount = 0;
|
|
108
|
+
#requestSeenCache;
|
|
109
|
+
#queuePausedForMigration = false;
|
|
110
|
+
#inProgressRequestBatchCount = 0;
|
|
69
111
|
/**
|
|
70
112
|
* The largest expected request-processing time (in seconds) seen so far via
|
|
71
113
|
* {@link setExpectedRequestProcessingTimeSecs}. Used to ensure that value is only ever raised, never
|
|
72
114
|
* lowered, before being forwarded to the storage backend.
|
|
73
115
|
*/
|
|
74
|
-
expectedRequestProcessingSecs = 0;
|
|
75
|
-
httpClient;
|
|
76
|
-
events;
|
|
77
|
-
statsTracker = new StorageStatsTracker({
|
|
116
|
+
#expectedRequestProcessingSecs = 0;
|
|
117
|
+
#httpClient;
|
|
118
|
+
#events;
|
|
119
|
+
#statsTracker = new StorageStatsTracker({
|
|
78
120
|
writeCount: 0,
|
|
79
121
|
headItemReadCount: 0,
|
|
80
122
|
});
|
|
@@ -83,7 +125,7 @@ export class RequestQueue {
|
|
|
83
125
|
* queue-head reads issued to the underlying storage backend). Counted per backend call.
|
|
84
126
|
*/
|
|
85
127
|
get stats() {
|
|
86
|
-
return this
|
|
128
|
+
return this.#statsTracker.current;
|
|
87
129
|
}
|
|
88
130
|
/**
|
|
89
131
|
* @internal
|
|
@@ -91,14 +133,14 @@ export class RequestQueue {
|
|
|
91
133
|
constructor(options) {
|
|
92
134
|
this.id = options.metadata.id;
|
|
93
135
|
this.name = options.metadata.name;
|
|
94
|
-
this
|
|
136
|
+
this.#events = serviceLocator.getEventManager();
|
|
95
137
|
this.backend = options.backend;
|
|
96
|
-
this
|
|
97
|
-
this
|
|
98
|
-
this
|
|
138
|
+
this.#proxyConfiguration = options.proxyConfiguration;
|
|
139
|
+
this.#requestCache = new LruCache({ maxLength: MAX_CACHED_REQUESTS });
|
|
140
|
+
this.#requestSeenCache = new RequestDeduplicationCache();
|
|
99
141
|
this.log = serviceLocator.getLogger().child({ prefix: `RequestQueue(${this.id}, ${this.name ?? 'no-name'})` });
|
|
100
|
-
this
|
|
101
|
-
this
|
|
142
|
+
this.#events.on(EventType.MIGRATING, async () => {
|
|
143
|
+
this.#queuePausedForMigration = true;
|
|
102
144
|
});
|
|
103
145
|
}
|
|
104
146
|
/**
|
|
@@ -134,26 +176,24 @@ export class RequestQueue {
|
|
|
134
176
|
* @param [options] Request queue operation options.
|
|
135
177
|
*/
|
|
136
178
|
async addRequest(requestLike, options = {}) {
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
forefront: ow.optional.boolean,
|
|
141
|
-
}));
|
|
142
|
-
const { forefront = false } = options;
|
|
179
|
+
const transaction = activeStorageTransaction();
|
|
180
|
+
parseArgument(requestLike, schemas.anyObject);
|
|
181
|
+
const { forefront } = parseArgument(options, operationOptionsSchema);
|
|
143
182
|
if ('requestsFromUrl' in requestLike) {
|
|
144
|
-
const requests = await this
|
|
145
|
-
const processedRequests = await this
|
|
183
|
+
const requests = await this.#fetchRequestsFromUrl(requestLike);
|
|
184
|
+
const processedRequests = await this.#addFetchedRequests(requestLike, requests, options);
|
|
146
185
|
return { ...processedRequests[0], forefront };
|
|
147
186
|
}
|
|
148
|
-
|
|
149
|
-
url: ow.string,
|
|
150
|
-
id: ow.undefined,
|
|
151
|
-
}));
|
|
187
|
+
parseArgument(requestLike, newRequestLikeSchema);
|
|
152
188
|
const request = requestLike instanceof Request ? requestLike : new Request(requestLike);
|
|
189
|
+
if (transaction?.policy.requestQueue === 'deferred') {
|
|
190
|
+
return this.#addRequestDeferred(transaction, request, forefront);
|
|
191
|
+
}
|
|
153
192
|
const cacheKey = getRequestId(request.uniqueKey);
|
|
154
|
-
const cachedInfo = this
|
|
193
|
+
const cachedInfo = this.#requestCache.get(cacheKey);
|
|
155
194
|
if (cachedInfo) {
|
|
156
195
|
request.id = cachedInfo.id;
|
|
196
|
+
this.#recordRequestJournalEntry(transaction, [request], forefront, true);
|
|
157
197
|
return {
|
|
158
198
|
wasAlreadyPresent: true,
|
|
159
199
|
// We may assume that if request is in local cache then also the information if the
|
|
@@ -164,17 +204,163 @@ export class RequestQueue {
|
|
|
164
204
|
forefront,
|
|
165
205
|
};
|
|
166
206
|
}
|
|
167
|
-
this
|
|
207
|
+
this.#statsTracker.add('writeCount');
|
|
168
208
|
const { processedRequests } = await this.backend.addBatchOfRequests([request], { forefront });
|
|
209
|
+
this.#recordRequestJournalEntry(transaction, [request], forefront, true);
|
|
169
210
|
const queueOperationInfo = {
|
|
170
211
|
...processedRequests[0],
|
|
171
212
|
uniqueKey: request.uniqueKey,
|
|
172
213
|
forefront,
|
|
173
214
|
};
|
|
174
|
-
this
|
|
175
|
-
this
|
|
215
|
+
this.#cacheRequest(cacheKey, queueOperationInfo);
|
|
216
|
+
this.#requestSeenCache.add(cacheKey, request.id);
|
|
176
217
|
return queueOperationInfo;
|
|
177
218
|
}
|
|
219
|
+
/**
|
|
220
|
+
* Journals an addition for introspection only; these entries are never replayed. A no-op unless the
|
|
221
|
+
* transaction is open, so detached and outliving writers stay out of the journal.
|
|
222
|
+
*/
|
|
223
|
+
#recordRequestJournalEntry(transaction, requests, forefront, writeThrough) {
|
|
224
|
+
if (!transaction?.isActive || requests.length === 0)
|
|
225
|
+
return;
|
|
226
|
+
transaction.recordJournalEntry({
|
|
227
|
+
type: 'requestQueue',
|
|
228
|
+
participant: this,
|
|
229
|
+
requests: requests.map((request) => ({
|
|
230
|
+
url: request.url,
|
|
231
|
+
uniqueKey: request.uniqueKey,
|
|
232
|
+
label: request.label,
|
|
233
|
+
})),
|
|
234
|
+
forefront,
|
|
235
|
+
writeThrough,
|
|
236
|
+
});
|
|
237
|
+
}
|
|
238
|
+
/**
|
|
239
|
+
* The requests buffered by the given transaction for this queue, keyed by `uniqueKey` — a dedup
|
|
240
|
+
* index derived from the transaction journal.
|
|
241
|
+
*/
|
|
242
|
+
#bufferedRequests(transaction) {
|
|
243
|
+
const buffered = new Map();
|
|
244
|
+
// Only `deferred` records snapshots, so scanning the journal under `writeThrough` never finds any.
|
|
245
|
+
if (transaction.policy.requestQueue !== 'deferred')
|
|
246
|
+
return buffered;
|
|
247
|
+
for (const entry of transaction.journal) {
|
|
248
|
+
if (entry.type !== 'requestQueue' || entry.participant !== this)
|
|
249
|
+
continue;
|
|
250
|
+
for (const request of entry.requests) {
|
|
251
|
+
if (request.snapshot !== undefined)
|
|
252
|
+
buffered.set(request.uniqueKey, request.snapshot);
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
return buffered;
|
|
256
|
+
}
|
|
257
|
+
/**
|
|
258
|
+
* Adds a request under the `deferred` policy: journaled now, really added by the commit replay.
|
|
259
|
+
* A new request's `requestId` is the local `uniqueKey` hash and is **provisional** — never write it
|
|
260
|
+
* to `request.id` or the dedup caches. Dedup is cheapest-first: buffer, caches, then a backend probe.
|
|
261
|
+
*/
|
|
262
|
+
async #addRequestDeferred(transaction, request, forefront, buffered = this.#bufferedRequests(transaction)) {
|
|
263
|
+
// This transaction's own buffered adds; the shared caches never see them (provisional ids).
|
|
264
|
+
if (buffered.has(request.uniqueKey)) {
|
|
265
|
+
this.#recordRequestJournalEntry(transaction, [request], forefront, false);
|
|
266
|
+
return {
|
|
267
|
+
wasAlreadyPresent: true,
|
|
268
|
+
wasAlreadyHandled: false,
|
|
269
|
+
requestId: getRequestId(request.uniqueKey),
|
|
270
|
+
uniqueKey: request.uniqueKey,
|
|
271
|
+
forefront,
|
|
272
|
+
};
|
|
273
|
+
}
|
|
274
|
+
// The caches hold real backend ids. Only *writing* provisional ids to them would be wrong;
|
|
275
|
+
// reading saves a probe. Same lookup as the write-through path.
|
|
276
|
+
const cacheKey = getRequestId(request.uniqueKey);
|
|
277
|
+
const cachedInfo = this.#requestCache.get(cacheKey);
|
|
278
|
+
const knownRequestId = cachedInfo?.id ?? this.#requestSeenCache.get(cacheKey);
|
|
279
|
+
if (knownRequestId) {
|
|
280
|
+
this.#recordRequestJournalEntry(transaction, [request], forefront, false);
|
|
281
|
+
return {
|
|
282
|
+
wasAlreadyPresent: true,
|
|
283
|
+
// The dedup cache doesn't track the handled state; only the full record does.
|
|
284
|
+
wasAlreadyHandled: cachedInfo?.isHandled ?? false,
|
|
285
|
+
requestId: knownRequestId,
|
|
286
|
+
uniqueKey: request.uniqueKey,
|
|
287
|
+
forefront,
|
|
288
|
+
};
|
|
289
|
+
}
|
|
290
|
+
// The caches are bounded, so a miss is not proof of absence - probe for an accurate answer.
|
|
291
|
+
const existing = await this.backend.getRequest(request.uniqueKey);
|
|
292
|
+
if (existing) {
|
|
293
|
+
this.#recordRequestJournalEntry(transaction, [request], forefront, false);
|
|
294
|
+
return {
|
|
295
|
+
wasAlreadyPresent: true,
|
|
296
|
+
wasAlreadyHandled: existing.handledAt != null,
|
|
297
|
+
requestId: existing.id,
|
|
298
|
+
uniqueKey: request.uniqueKey,
|
|
299
|
+
forefront,
|
|
300
|
+
};
|
|
301
|
+
}
|
|
302
|
+
// The entry below *is* the write, so a transaction closed during the probe must not receive it -
|
|
303
|
+
// pass through instead, per the closed-transaction rule. Under `deferred` that can land an
|
|
304
|
+
// addition a rollback would have discarded; dedup bounds that cost, silent loss is unbounded.
|
|
305
|
+
if (!transaction.isActive) {
|
|
306
|
+
return await this.addRequest(request, { forefront });
|
|
307
|
+
}
|
|
308
|
+
const snapshot = JSON.parse(JSON.stringify(request));
|
|
309
|
+
// Strip-list, not allow-list: every user-facing field flows through, including ones added to
|
|
310
|
+
// `Request` in the future. The exceptions are `id` and `handledAt`, the two backend-owned
|
|
311
|
+
// lifecycle fields.
|
|
312
|
+
delete snapshot.id;
|
|
313
|
+
delete snapshot.handledAt;
|
|
314
|
+
transaction.recordJournalEntry({
|
|
315
|
+
type: 'requestQueue',
|
|
316
|
+
participant: this,
|
|
317
|
+
requests: [{ url: request.url, uniqueKey: request.uniqueKey, label: request.label, snapshot }],
|
|
318
|
+
forefront,
|
|
319
|
+
writeThrough: false,
|
|
320
|
+
});
|
|
321
|
+
buffered.set(request.uniqueKey, snapshot);
|
|
322
|
+
return {
|
|
323
|
+
wasAlreadyPresent: false,
|
|
324
|
+
wasAlreadyHandled: false,
|
|
325
|
+
requestId: getRequestId(request.uniqueKey),
|
|
326
|
+
uniqueKey: request.uniqueKey,
|
|
327
|
+
forefront,
|
|
328
|
+
};
|
|
329
|
+
}
|
|
330
|
+
/** @internal */
|
|
331
|
+
async commitJournalEntries(entries) {
|
|
332
|
+
// Replay through `backend.addBatchOfRequests`, *not* the batched frontend wrapper - the wrapper
|
|
333
|
+
// resolves after the first chunk and sleeps between the rest, neither of which commit may
|
|
334
|
+
// inherit. One call per `forefront` flag; the order of forefront additions is arbitrary anyway.
|
|
335
|
+
for (const forefront of [false, true]) {
|
|
336
|
+
const requests = entries.flatMap((entry) => entry.type === 'requestQueue' && entry.forefront === forefront
|
|
337
|
+
? // Requests without a snapshot were deduplicated or written through; nothing to replay.
|
|
338
|
+
entry.requests
|
|
339
|
+
.filter((journaled) => journaled.snapshot !== undefined)
|
|
340
|
+
.map((journaled) => Request.fromSchema(journaled.snapshot))
|
|
341
|
+
: []);
|
|
342
|
+
if (requests.length === 0)
|
|
343
|
+
continue;
|
|
344
|
+
this.#statsTracker.add('writeCount');
|
|
345
|
+
const { processedRequests, unprocessedRequests } = await this.backend.addBatchOfRequests(requests, {
|
|
346
|
+
forefront,
|
|
347
|
+
});
|
|
348
|
+
// Only now, with the real backend-assigned ids, may the shared dedup caches be populated.
|
|
349
|
+
for (const processed of processedRequests) {
|
|
350
|
+
const cacheKey = getRequestId(processed.uniqueKey);
|
|
351
|
+
this.#cacheRequest(cacheKey, { ...processed, forefront });
|
|
352
|
+
this.#requestSeenCache.add(cacheKey, processed.requestId);
|
|
353
|
+
}
|
|
354
|
+
if (unprocessedRequests.length > 0) {
|
|
355
|
+
// Warn and skip, rather than retry or fail. `unprocessedRequests` is what remains after
|
|
356
|
+
// the backend's own transient-error handling - a semantic rejection that retrying here
|
|
357
|
+
// would only re-poke. And failing the commit would let one malformed request hold the
|
|
358
|
+
// whole transaction hostage.
|
|
359
|
+
this.log.warning('Some requests were rejected by the request queue while committing a storage transaction and will be skipped. ' +
|
|
360
|
+
"This usually means the request data is malformed (e.g. an invalid 'userData' shape).", { unprocessedRequests });
|
|
361
|
+
}
|
|
362
|
+
}
|
|
363
|
+
}
|
|
178
364
|
/**
|
|
179
365
|
* Adds requests to the queue in batches of 25. This method will wait till all the requests are added
|
|
180
366
|
* to the queue before resolving. You should prefer using `queue.addRequestsBatched()` or `crawler.addRequests()`
|
|
@@ -190,15 +376,9 @@ export class RequestQueue {
|
|
|
190
376
|
* @param [options] Request queue operation options.
|
|
191
377
|
*/
|
|
192
378
|
async addRequests(requestsLike, options = {}) {
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
.message((value) => `Expected an iterable or async iterable, got ${getObjectType(value)}`));
|
|
197
|
-
ow(options, ow.object.exactShape({
|
|
198
|
-
forefront: ow.optional.boolean,
|
|
199
|
-
cache: ow.optional.boolean,
|
|
200
|
-
}));
|
|
201
|
-
const { forefront = false, cache = true } = options;
|
|
379
|
+
const transaction = activeStorageTransaction();
|
|
380
|
+
parseArgument(requestsLike, iterableSchema);
|
|
381
|
+
const { forefront, cache } = parseArgument(options, addRequestsOptionsSchema);
|
|
202
382
|
const uniqueKeyToCacheKey = new Map();
|
|
203
383
|
const getCachedRequestId = (uniqueKey) => {
|
|
204
384
|
const cached = uniqueKeyToCacheKey.get(uniqueKey);
|
|
@@ -218,19 +398,27 @@ export class RequestQueue {
|
|
|
218
398
|
requests.push(new Request({ url: requestLike }));
|
|
219
399
|
}
|
|
220
400
|
else if ('requestsFromUrl' in requestLike) {
|
|
221
|
-
const fetchedRequests = await this
|
|
222
|
-
await this
|
|
401
|
+
const fetchedRequests = await this.#fetchRequestsFromUrl(requestLike);
|
|
402
|
+
await this.#addFetchedRequests(requestLike, fetchedRequests, options);
|
|
223
403
|
}
|
|
224
404
|
else {
|
|
225
405
|
requests.push(requestLike instanceof Request ? requestLike : new Request(requestLike));
|
|
226
406
|
}
|
|
227
407
|
}
|
|
408
|
+
if (transaction?.policy.requestQueue === 'deferred') {
|
|
409
|
+
const buffered = this.#bufferedRequests(transaction);
|
|
410
|
+
for (const request of requests) {
|
|
411
|
+
results.processedRequests.push(await this.#addRequestDeferred(transaction, request, forefront, buffered));
|
|
412
|
+
}
|
|
413
|
+
return results;
|
|
414
|
+
}
|
|
415
|
+
this.#recordRequestJournalEntry(transaction, requests, forefront, true);
|
|
228
416
|
const requestsToAdd = new Map();
|
|
229
417
|
for (const request of requests) {
|
|
230
418
|
const cacheKey = getCachedRequestId(request.uniqueKey);
|
|
231
419
|
// Prefer the full `requestCache` record; fall back to the dedup cache for background batches it skips.
|
|
232
|
-
const cachedInfo = this
|
|
233
|
-
const knownRequestId = cachedInfo?.id ?? this
|
|
420
|
+
const cachedInfo = this.#requestCache.get(cacheKey);
|
|
421
|
+
const knownRequestId = cachedInfo?.id ?? this.#requestSeenCache.get(cacheKey);
|
|
234
422
|
if (knownRequestId) {
|
|
235
423
|
request.id = knownRequestId;
|
|
236
424
|
results.processedRequests.push({
|
|
@@ -249,7 +437,7 @@ export class RequestQueue {
|
|
|
249
437
|
if (!requestsToAdd.size) {
|
|
250
438
|
return results;
|
|
251
439
|
}
|
|
252
|
-
this
|
|
440
|
+
this.#statsTracker.add('writeCount');
|
|
253
441
|
const apiResults = await this.backend.addBatchOfRequests([...requestsToAdd.values()], { forefront });
|
|
254
442
|
// Report unprocessed requests
|
|
255
443
|
results.unprocessedRequests = apiResults.unprocessedRequests;
|
|
@@ -259,10 +447,10 @@ export class RequestQueue {
|
|
|
259
447
|
results.processedRequests.push(newRequest);
|
|
260
448
|
const cacheKey = getCachedRequestId(newRequest.uniqueKey);
|
|
261
449
|
if (cache) {
|
|
262
|
-
this
|
|
450
|
+
this.#cacheRequest(cacheKey, { ...newRequest, forefront });
|
|
263
451
|
}
|
|
264
452
|
// Unlike `requestCache`, populate this on every batch (including background ones).
|
|
265
|
-
this
|
|
453
|
+
this.#requestSeenCache.add(cacheKey, newRequest.requestId);
|
|
266
454
|
}
|
|
267
455
|
return results;
|
|
268
456
|
}
|
|
@@ -276,17 +464,8 @@ export class RequestQueue {
|
|
|
276
464
|
* @param options Options for the request queue
|
|
277
465
|
*/
|
|
278
466
|
async addRequestsBatched(requests, options = {}) {
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
.is((value) => isIterable(value) || isAsyncIterable(value))
|
|
282
|
-
.message((value) => `Expected an iterable or async iterable, got ${getObjectType(value)}`));
|
|
283
|
-
ow(options, ow.object.exactShape({
|
|
284
|
-
forefront: ow.optional.boolean,
|
|
285
|
-
waitForAllRequestsToBeAdded: ow.optional.boolean,
|
|
286
|
-
batchSize: ow.optional.number,
|
|
287
|
-
waitBetweenBatchesMillis: ow.optional.number,
|
|
288
|
-
maxNewRequests: ow.optional.number,
|
|
289
|
-
}));
|
|
467
|
+
parseArgument(requests, iterableSchema);
|
|
468
|
+
const { forefront, waitForAllRequestsToBeAdded, batchSize, waitBetweenBatchesMillis, maxNewRequests } = parseArgument(options, addRequestsBatchedOptionsSchema);
|
|
290
469
|
const addRequest = this.addRequest.bind(this);
|
|
291
470
|
async function* generateRequests() {
|
|
292
471
|
for await (const opts of requests) {
|
|
@@ -295,7 +474,7 @@ export class RequestQueue {
|
|
|
295
474
|
if (opts.url !== undefined && typeof opts.url !== 'string') {
|
|
296
475
|
throw new Error(`Request options are not valid, the 'url' property is not a string. Input: ${inspect(opts)}`);
|
|
297
476
|
}
|
|
298
|
-
if (opts.id !== undefined) {
|
|
477
|
+
if ('id' in opts && opts.id !== undefined) {
|
|
299
478
|
throw new Error(`Request options are not valid, the 'id' property must not be present. Input: ${inspect(opts)}`);
|
|
300
479
|
}
|
|
301
480
|
if (opts.requestsFromUrl !== undefined &&
|
|
@@ -305,7 +484,7 @@ export class RequestQueue {
|
|
|
305
484
|
}
|
|
306
485
|
if (opts && typeof opts === 'object' && 'requestsFromUrl' in opts) {
|
|
307
486
|
// Handle URL lists right away
|
|
308
|
-
await addRequest(opts, { forefront
|
|
487
|
+
await addRequest(opts, { forefront });
|
|
309
488
|
}
|
|
310
489
|
else {
|
|
311
490
|
// Yield valid requests
|
|
@@ -313,84 +492,36 @@ export class RequestQueue {
|
|
|
313
492
|
}
|
|
314
493
|
}
|
|
315
494
|
}
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
cache,
|
|
337
|
-
});
|
|
338
|
-
if (unprocessedRequests.length > 0) {
|
|
339
|
-
this.log.warning('Some requests were rejected by the request queue and will be skipped. ' +
|
|
340
|
-
"This usually means the request data is malformed (e.g. an invalid 'userData' shape).", { unprocessedRequests });
|
|
341
|
-
}
|
|
342
|
-
if (maxNewRequests !== undefined) {
|
|
343
|
-
remainingBudget -= processedRequests.filter((r) => !r.wasAlreadyPresent).length;
|
|
344
|
-
}
|
|
345
|
-
return processedRequests;
|
|
346
|
-
};
|
|
347
|
-
/**
|
|
348
|
-
* Build the final result. When maxNewRequests is set, drains any remaining items
|
|
349
|
-
* from the underlying request iterator into requestsOverLimit.
|
|
350
|
-
*
|
|
351
|
-
* We accept the iterator explicitly (rather than closing over it) to make it obvious
|
|
352
|
-
* that this is the *same* iterator that `chunkedAsyncIterable` has been consuming —
|
|
353
|
-
* so only unconsumed items are drained. We drain `requestIterator` (not `chunks`)
|
|
354
|
-
* because `chunkedAsyncIterable` stops yielding when the budget-based chunk size
|
|
355
|
-
* drops to 0, leaving unconsumed items in the underlying iterator.
|
|
356
|
-
*/
|
|
357
|
-
const buildResult = async (addedRequests, waitForAllRequestsToBeAdded, unconsumedIterator) => {
|
|
358
|
-
if (maxNewRequests !== undefined) {
|
|
359
|
-
for await (const request of unconsumedIterator) {
|
|
360
|
-
requestsOverLimit.push(request);
|
|
495
|
+
return drainRequestBatches({
|
|
496
|
+
items: generateRequests(),
|
|
497
|
+
batchSize,
|
|
498
|
+
waitBetweenBatchesMillis,
|
|
499
|
+
waitForAllRequestsToBeAdded,
|
|
500
|
+
maxNewRequests,
|
|
501
|
+
/**
|
|
502
|
+
* Requests the backend reports as unprocessed are warned about and skipped rather than retried:
|
|
503
|
+
* `unprocessedRequests` is what remains after the backend's own transient-error handling - a
|
|
504
|
+
* semantic rejection (e.g. a malformed `userData` shape) that re-sending would only re-poke.
|
|
505
|
+
* Retrying transient failures is the storage backend's job, not the frontend's.
|
|
506
|
+
*/
|
|
507
|
+
processChunk: async (chunk, isInitial) => {
|
|
508
|
+
const { processedRequests, unprocessedRequests } = await this.addRequests(chunk, {
|
|
509
|
+
forefront,
|
|
510
|
+
cache: isInitial,
|
|
511
|
+
});
|
|
512
|
+
if (unprocessedRequests.length > 0) {
|
|
513
|
+
this.log.warning('Some requests were rejected by the request queue and will be skipped. ' +
|
|
514
|
+
"This usually means the request data is malformed (e.g. an invalid 'userData' shape).", { unprocessedRequests });
|
|
361
515
|
}
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
const addedRequests = await processChunk(initialChunk);
|
|
371
|
-
await chunksIterator.next();
|
|
372
|
-
// If we have no more requests to add (either exhausted or budget hit), return immediately
|
|
373
|
-
if ((await chunksIterator.peek()) === undefined) {
|
|
374
|
-
return buildResult(addedRequests, Promise.resolve([]), requestIterator);
|
|
375
|
-
}
|
|
376
|
-
// eslint-disable-next-line no-async-promise-executor
|
|
377
|
-
const promise = new Promise(async (resolve) => {
|
|
378
|
-
const finalAddedRequests = [];
|
|
379
|
-
for await (const requestChunk of chunks) {
|
|
380
|
-
finalAddedRequests.push(...(await processChunk(requestChunk, false)));
|
|
381
|
-
await sleep(waitBetweenBatchesMillis);
|
|
382
|
-
}
|
|
383
|
-
resolve(finalAddedRequests);
|
|
384
|
-
});
|
|
385
|
-
this.inProgressRequestBatchCount += 1;
|
|
386
|
-
void promise.finally(() => {
|
|
387
|
-
this.inProgressRequestBatchCount -= 1;
|
|
516
|
+
return processedRequests;
|
|
517
|
+
},
|
|
518
|
+
trackBackgroundBatches: (batches) => {
|
|
519
|
+
this.#inProgressRequestBatchCount += 1;
|
|
520
|
+
void batches.finally(() => {
|
|
521
|
+
this.#inProgressRequestBatchCount -= 1;
|
|
522
|
+
});
|
|
523
|
+
},
|
|
388
524
|
});
|
|
389
|
-
// When maxNewRequests is set, we must wait for all batches so we can accurately report skipped requests.
|
|
390
|
-
if (options.waitForAllRequestsToBeAdded || maxNewRequests !== undefined) {
|
|
391
|
-
addedRequests.push(...(await promise));
|
|
392
|
-
}
|
|
393
|
-
return buildResult(addedRequests, promise, requestIterator);
|
|
394
525
|
}
|
|
395
526
|
/**
|
|
396
527
|
* Gets the request from the queue specified by its `uniqueKey`.
|
|
@@ -399,12 +530,17 @@ export class RequestQueue {
|
|
|
399
530
|
* @returns Returns the request object, or `null` if it was not found.
|
|
400
531
|
*/
|
|
401
532
|
async getRequest(uniqueKey) {
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
533
|
+
const transaction = activeStorageTransaction();
|
|
534
|
+
parseArgument(uniqueKey, uniqueKeySchema);
|
|
535
|
+
// Requests buffered by the active transaction (under the `deferred` write policy) are visible to it.
|
|
536
|
+
const buffered = transaction && this.#bufferedRequests(transaction).get(uniqueKey);
|
|
537
|
+
if (buffered) {
|
|
538
|
+
return Request.fromSchema(buffered);
|
|
539
|
+
}
|
|
540
|
+
const schema = await this.backend.getRequest(uniqueKey);
|
|
541
|
+
if (!schema)
|
|
406
542
|
return null;
|
|
407
|
-
return
|
|
543
|
+
return Request.fromSchema(schema);
|
|
408
544
|
}
|
|
409
545
|
/**
|
|
410
546
|
* Returns a next request in the queue to be processed, or `null` if there are no more pending requests.
|
|
@@ -418,21 +554,21 @@ export class RequestQueue {
|
|
|
418
554
|
* Note that the `null` return value doesn't mean the queue processing finished,
|
|
419
555
|
* it means there are currently no pending requests.
|
|
420
556
|
* To check whether all requests in queue were finished,
|
|
421
|
-
* use {@link RequestQueue.
|
|
557
|
+
* use {@link RequestQueue.checkReadiness} instead.
|
|
422
558
|
*
|
|
423
559
|
* @returns
|
|
424
560
|
* Returns the request object or `null` if there are no more pending requests.
|
|
425
561
|
*/
|
|
426
562
|
async fetchNextRequest() {
|
|
427
|
-
|
|
428
|
-
if (this
|
|
563
|
+
rejectOperationInTransaction('RequestQueue.fetchNextRequest()', 'it is part of the crawler request-processing bookkeeping, which a transaction must not affect.');
|
|
564
|
+
if (this.#queuePausedForMigration) {
|
|
429
565
|
return null;
|
|
430
566
|
}
|
|
431
|
-
this
|
|
432
|
-
const
|
|
433
|
-
if (!
|
|
567
|
+
this.#statsTracker.add('headItemReadCount');
|
|
568
|
+
const schema = await this.backend.fetchNextRequest();
|
|
569
|
+
if (!schema)
|
|
434
570
|
return null;
|
|
435
|
-
return
|
|
571
|
+
return Request.fromSchema(schema);
|
|
436
572
|
}
|
|
437
573
|
/**
|
|
438
574
|
* Marks a request that was previously returned by the
|
|
@@ -441,15 +577,11 @@ export class RequestQueue {
|
|
|
441
577
|
* Handled requests will never again be returned by the `fetchNextRequest` function.
|
|
442
578
|
*/
|
|
443
579
|
async markRequestAsHandled(request) {
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
uniqueKey: ow.string,
|
|
448
|
-
handledAt: ow.optional.string,
|
|
449
|
-
}));
|
|
450
|
-
const forefront = this.requestCache.get(getRequestId(request.uniqueKey))?.forefront ?? false;
|
|
580
|
+
rejectOperationInTransaction('RequestQueue.markRequestAsHandled()', 'it is part of the crawler request-processing bookkeeping, which a transaction must not affect.');
|
|
581
|
+
parseArgument(request, handledRequestSchema);
|
|
582
|
+
const forefront = this.#requestCache.get(getRequestId(request.uniqueKey))?.forefront ?? false;
|
|
451
583
|
const handledAt = request.handledAt ?? new Date().toISOString();
|
|
452
|
-
this
|
|
584
|
+
this.#statsTracker.add('writeCount');
|
|
453
585
|
const processedRequest = await this.backend.markRequestAsHandled({
|
|
454
586
|
...request,
|
|
455
587
|
handledAt,
|
|
@@ -464,7 +596,7 @@ export class RequestQueue {
|
|
|
464
596
|
uniqueKey: request.uniqueKey,
|
|
465
597
|
forefront,
|
|
466
598
|
};
|
|
467
|
-
this
|
|
599
|
+
this.#cacheRequest(getRequestId(request.uniqueKey), queueOperationInfo);
|
|
468
600
|
return queueOperationInfo;
|
|
469
601
|
}
|
|
470
602
|
/**
|
|
@@ -474,17 +606,13 @@ export class RequestQueue {
|
|
|
474
606
|
* For example, this lets you store the number of retries or error messages for the request.
|
|
475
607
|
*/
|
|
476
608
|
async reclaimRequest(request, options = {}) {
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
}));
|
|
485
|
-
const { forefront = false } = options;
|
|
486
|
-
this.statsTracker.add('writeCount');
|
|
487
|
-
const processedRequest = await this.backend.reclaimRequest(request, { forefront });
|
|
609
|
+
rejectOperationInTransaction('RequestQueue.reclaimRequest()', 'it is part of the crawler request-processing bookkeeping, which a transaction must not affect.');
|
|
610
|
+
parseArgument(request, reclaimedRequestSchema);
|
|
611
|
+
const { forefront } = parseArgument(options, operationOptionsSchema);
|
|
612
|
+
this.#statsTracker.add('writeCount');
|
|
613
|
+
const processedRequest = await this.backend.reclaimRequest(request, {
|
|
614
|
+
forefront,
|
|
615
|
+
});
|
|
488
616
|
// The request was not in progress — nothing to reclaim.
|
|
489
617
|
if (!processedRequest) {
|
|
490
618
|
return null;
|
|
@@ -494,38 +622,41 @@ export class RequestQueue {
|
|
|
494
622
|
uniqueKey: request.uniqueKey,
|
|
495
623
|
forefront,
|
|
496
624
|
};
|
|
497
|
-
this
|
|
625
|
+
this.#cacheRequest(getRequestId(request.uniqueKey), queueOperationInfo);
|
|
498
626
|
return queueOperationInfo;
|
|
499
627
|
}
|
|
500
628
|
/**
|
|
501
|
-
*
|
|
502
|
-
*
|
|
503
|
-
*
|
|
504
|
-
* Note that even if the queue is empty, there might be some requests currently being processed
|
|
505
|
-
* (fetched but not yet handled or reclaimed). An empty queue therefore does not mean crawling is
|
|
506
|
-
* finished — those in-progress requests may still be reclaimed, and background tasks may still be
|
|
507
|
-
* adding more requests. To check whether all activity in the queue has finished, use
|
|
508
|
-
* {@link RequestQueue.isFinished}.
|
|
629
|
+
* A queue hands requests out as fast as they are asked for; pacing is a job for a manager wrapped around it,
|
|
630
|
+
* such as {@link ThrottlingRequestManager}.
|
|
631
|
+
* @inheritdoc
|
|
509
632
|
*/
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
return this.backend.isEmpty();
|
|
633
|
+
recordPacingSignal(_signal) {
|
|
634
|
+
return false;
|
|
513
635
|
}
|
|
514
636
|
/**
|
|
515
|
-
*
|
|
516
|
-
*
|
|
517
|
-
*
|
|
637
|
+
* Reports whether the queue has a request to hand over, is waiting on one, or is done.
|
|
638
|
+
*
|
|
639
|
+
* `waiting` means requests are in progress (fetched but not yet handled or reclaimed, possibly by another
|
|
640
|
+
* client sharing the queue) or a background add is still landing; neither has a clock, so no `readyAt`.
|
|
518
641
|
*
|
|
519
|
-
* Due to the nature of distributed storage used by the queue,
|
|
520
|
-
*
|
|
642
|
+
* Due to the nature of distributed storage used by the queue, `finished` may occasionally arrive a probe or
|
|
643
|
+
* two late, but it is never reported early.
|
|
521
644
|
*/
|
|
522
|
-
async
|
|
523
|
-
|
|
645
|
+
async checkReadiness() {
|
|
646
|
+
const transaction = activeStorageTransaction();
|
|
647
|
+
// Requests buffered by the active transaction count as pending from its point of view.
|
|
648
|
+
if (transaction && this.#bufferedRequests(transaction).size > 0) {
|
|
649
|
+
return { status: 'ready' };
|
|
650
|
+
}
|
|
651
|
+
// Something fetchable outranks everything below, so this is the only backend call a probe needs.
|
|
652
|
+
if (!(await this.backend.isEmpty())) {
|
|
653
|
+
return { status: 'ready' };
|
|
654
|
+
}
|
|
524
655
|
// We are not finished if we're still adding new requests in the background.
|
|
525
|
-
if (this
|
|
526
|
-
return
|
|
656
|
+
if (this.#inProgressRequestBatchCount > 0) {
|
|
657
|
+
return { status: 'waiting' };
|
|
527
658
|
}
|
|
528
|
-
return this.backend.isFinished();
|
|
659
|
+
return (await this.backend.isFinished()) ? { status: 'finished' } : { status: 'waiting' };
|
|
529
660
|
}
|
|
530
661
|
/**
|
|
531
662
|
* Tells the queue how long a consumer expects to hold a fetched request before marking it handled
|
|
@@ -538,24 +669,34 @@ export class RequestQueue {
|
|
|
538
669
|
* short the reservation of a long-lived one and have its in-flight request stolen.
|
|
539
670
|
*/
|
|
540
671
|
async setExpectedRequestProcessingTimeSecs(secs) {
|
|
541
|
-
if (secs <= this
|
|
672
|
+
if (secs <= this.#expectedRequestProcessingSecs) {
|
|
542
673
|
return;
|
|
543
674
|
}
|
|
544
|
-
this
|
|
675
|
+
this.#expectedRequestProcessingSecs = secs;
|
|
545
676
|
await this.backend.setExpectedRequestProcessingTimeSecs?.(secs);
|
|
546
677
|
}
|
|
678
|
+
/**
|
|
679
|
+
* @inheritdoc
|
|
680
|
+
* Unlike {@link RequestQueue.setExpectedRequestProcessingTimeSecs}, which sizes every future lock,
|
|
681
|
+
* this only touches the one request it is given.
|
|
682
|
+
*/
|
|
683
|
+
async extendRequestProcessingTimeSecs(request, secs) {
|
|
684
|
+
if (!request.id) {
|
|
685
|
+
return false;
|
|
686
|
+
}
|
|
687
|
+
return (await this.backend.extendRequestProcessingTimeSecs?.(request.id, secs)) ?? false;
|
|
688
|
+
}
|
|
547
689
|
/**
|
|
548
690
|
* Caches information about request to beware of unneeded addRequest() calls.
|
|
549
691
|
*/
|
|
550
|
-
cacheRequest(cacheKey, queueOperationInfo) {
|
|
692
|
+
#cacheRequest(cacheKey, queueOperationInfo) {
|
|
551
693
|
// Remove the previous entry, as otherwise our cache will never update 👀
|
|
552
|
-
this
|
|
553
|
-
this
|
|
694
|
+
this.#requestCache.remove(cacheKey);
|
|
695
|
+
this.#requestCache.add(cacheKey, {
|
|
554
696
|
id: queueOperationInfo.requestId,
|
|
555
697
|
isHandled: queueOperationInfo.wasAlreadyHandled,
|
|
556
698
|
uniqueKey: queueOperationInfo.uniqueKey,
|
|
557
699
|
hydrated: null,
|
|
558
|
-
lockExpiresAt: null,
|
|
559
700
|
forefront: queueOperationInfo.forefront,
|
|
560
701
|
});
|
|
561
702
|
}
|
|
@@ -564,7 +705,7 @@ export class RequestQueue {
|
|
|
564
705
|
* depending on the mode of operation.
|
|
565
706
|
*/
|
|
566
707
|
async drop() {
|
|
567
|
-
|
|
708
|
+
rejectOperationInTransaction('RequestQueue.drop()');
|
|
568
709
|
await this.backend.drop();
|
|
569
710
|
serviceLocator.getStorageInstanceManager().removeFromCache(this);
|
|
570
711
|
}
|
|
@@ -573,16 +714,16 @@ export class RequestQueue {
|
|
|
573
714
|
* so it can be reused (e.g. across multiple `crawler.run()` calls).
|
|
574
715
|
*/
|
|
575
716
|
async purge() {
|
|
576
|
-
|
|
717
|
+
rejectOperationInTransaction('RequestQueue.purge()');
|
|
577
718
|
await this.backend.purge();
|
|
578
719
|
// Reset in-memory bookkeeping so the queue behaves as if freshly opened.
|
|
579
|
-
this
|
|
580
|
-
this
|
|
581
|
-
this
|
|
720
|
+
this.#requestCache.clear();
|
|
721
|
+
this.#requestSeenCache.clear();
|
|
722
|
+
this.#inProgressRequestBatchCount = 0;
|
|
582
723
|
// Reset the expected-processing-time high-water mark too, otherwise the monotonic-raise guard
|
|
583
724
|
// in `setExpectedRequestProcessingTimeSecs` would let a value raised in an earlier run leak into a
|
|
584
725
|
// later one and silently swallow a lower hint (the queue is meant to be reusable across runs).
|
|
585
|
-
this
|
|
726
|
+
this.#expectedRequestProcessingSecs = 0;
|
|
586
727
|
}
|
|
587
728
|
/**
|
|
588
729
|
* @inheritdoc
|
|
@@ -630,21 +771,30 @@ export class RequestQueue {
|
|
|
630
771
|
* @throws If the underlying storage no longer exists (e.g. it was deleted externally).
|
|
631
772
|
*/
|
|
632
773
|
async getInfo() {
|
|
633
|
-
|
|
634
|
-
|
|
774
|
+
const transaction = activeStorageTransaction();
|
|
775
|
+
const metadata = await this.backend.getMetadata();
|
|
776
|
+
const bufferedCount = transaction ? this.#bufferedRequests(transaction).size : 0;
|
|
777
|
+
if (bufferedCount > 0) {
|
|
778
|
+
return {
|
|
779
|
+
...metadata,
|
|
780
|
+
totalRequestCount: metadata.totalRequestCount + bufferedCount,
|
|
781
|
+
pendingRequestCount: metadata.pendingRequestCount + bufferedCount,
|
|
782
|
+
};
|
|
783
|
+
}
|
|
784
|
+
return metadata;
|
|
635
785
|
}
|
|
636
786
|
/**
|
|
637
787
|
* Fetches URLs from requestsFromUrl and returns them in format of list of requests
|
|
638
788
|
*/
|
|
639
|
-
async fetchRequestsFromUrl(source) {
|
|
789
|
+
async #fetchRequestsFromUrl(source) {
|
|
640
790
|
const { requestsFromUrl, regex, ...sharedOpts } = source;
|
|
641
791
|
// Download remote resource and parse URLs.
|
|
642
792
|
let urlsArr;
|
|
643
793
|
try {
|
|
644
|
-
urlsArr = await this
|
|
794
|
+
urlsArr = await this.#downloadListOfUrls({
|
|
645
795
|
url: requestsFromUrl,
|
|
646
796
|
urlRegExp: regex,
|
|
647
|
-
proxyUrl: (await this
|
|
797
|
+
proxyUrl: (await this.#proxyConfiguration?.newProxyInfo())?.url,
|
|
648
798
|
});
|
|
649
799
|
}
|
|
650
800
|
catch (err) {
|
|
@@ -660,7 +810,7 @@ export class RequestQueue {
|
|
|
660
810
|
/**
|
|
661
811
|
* Adds all fetched requests from a URL from a remote resource.
|
|
662
812
|
*/
|
|
663
|
-
async addFetchedRequests(source, fetchedRequests, options) {
|
|
813
|
+
async #addFetchedRequests(source, fetchedRequests, options) {
|
|
664
814
|
const { requestsFromUrl, regex } = source;
|
|
665
815
|
const { addedRequests } = await this.addRequestsBatched(fetchedRequests, options);
|
|
666
816
|
this.log.info('Fetched and loaded Requests from a remote resource.', {
|
|
@@ -676,10 +826,10 @@ export class RequestQueue {
|
|
|
676
826
|
/**
|
|
677
827
|
* @internal wraps public utility for mocking purposes
|
|
678
828
|
*/
|
|
679
|
-
async
|
|
829
|
+
async #downloadListOfUrls(options) {
|
|
680
830
|
return downloadListOfUrls({
|
|
681
831
|
...options,
|
|
682
|
-
httpClient: this
|
|
832
|
+
httpClient: this.#httpClient,
|
|
683
833
|
});
|
|
684
834
|
}
|
|
685
835
|
/**
|
|
@@ -700,15 +850,10 @@ export class RequestQueue {
|
|
|
700
850
|
* @param [options] Open Request Queue options.
|
|
701
851
|
*/
|
|
702
852
|
static async open(identifier, options = {}) {
|
|
703
|
-
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
proxyConfiguration: ow.optional.object,
|
|
708
|
-
httpClient: ow.optional.object,
|
|
709
|
-
}));
|
|
710
|
-
const storageBackend = options.storageBackend ?? serviceLocator.getStorageBackend();
|
|
711
|
-
const configuration = options.configuration ?? serviceLocator.getConfiguration();
|
|
853
|
+
tryCancel();
|
|
854
|
+
const parsedOptions = parseArgument(options, openOptionsSchema);
|
|
855
|
+
const storageBackend = parsedOptions.storageBackend ?? serviceLocator.getStorageBackend();
|
|
856
|
+
const configuration = parsedOptions.configuration ?? serviceLocator.getConfiguration();
|
|
712
857
|
await purgeDefaultStorages({ onlyPurgeOnce: true, storageBackend, configuration });
|
|
713
858
|
const resolved = await resolveStorageIdentifier(identifier, storageBackend, 'RequestQueue');
|
|
714
859
|
const queue = await serviceLocator
|
|
@@ -718,8 +863,8 @@ export class RequestQueue {
|
|
|
718
863
|
backendOpener: () => storageBackend.createRequestQueueBackend(resolved),
|
|
719
864
|
backendCacheKey: storageBackend.getStorageBackendCacheKey?.() ?? storageBackend.constructor.name,
|
|
720
865
|
});
|
|
721
|
-
queue
|
|
722
|
-
queue
|
|
866
|
+
queue.#proxyConfiguration = parsedOptions.proxyConfiguration;
|
|
867
|
+
queue.#httpClient = parsedOptions.httpClient;
|
|
723
868
|
return queue;
|
|
724
869
|
}
|
|
725
870
|
}
|