@crawlee/core 4.0.0-beta.99 → 4.0.0-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. package/README.md +1 -1
  2. package/configuration.d.ts +16 -47
  3. package/configuration.js +13 -25
  4. package/debug.js +4 -4
  5. package/errors.d.ts +28 -38
  6. package/errors.js +33 -47
  7. package/events/event_manager.d.ts +2 -2
  8. package/events/event_manager.js +7 -6
  9. package/events/index.d.ts +1 -0
  10. package/events/local_event_manager.d.ts +1 -8
  11. package/events/local_event_manager.js +13 -13
  12. package/events/system_info.d.ts +38 -0
  13. package/index.d.ts +2 -8
  14. package/index.js +4 -8
  15. package/internal.d.ts +8 -0
  16. package/internal.js +9 -0
  17. package/log.d.ts +10 -11
  18. package/log.js +52 -20
  19. package/memory-storage/memory-storage.d.ts +15 -18
  20. package/memory-storage/memory-storage.js +80 -58
  21. package/memory-storage/resource-clients/dataset.d.ts +1 -6
  22. package/memory-storage/resource-clients/dataset.js +23 -31
  23. package/memory-storage/resource-clients/key-value-store.d.ts +1 -10
  24. package/memory-storage/resource-clients/key-value-store.js +43 -67
  25. package/memory-storage/resource-clients/request-queue.d.ts +1 -42
  26. package/memory-storage/resource-clients/request-queue.js +109 -117
  27. package/owned_or_injected.d.ts +1 -3
  28. package/owned_or_injected.js +17 -17
  29. package/package.json +17 -20
  30. package/proxy_configuration.d.ts +21 -26
  31. package/proxy_configuration.js +35 -25
  32. package/recoverable_state.d.ts +104 -47
  33. package/recoverable_state.js +199 -74
  34. package/request.d.ts +20 -107
  35. package/request.js +78 -244
  36. package/serialization.js +17 -16
  37. package/service_locator.d.ts +22 -10
  38. package/service_locator.js +59 -48
  39. package/storages/batched_adds.d.ts +37 -0
  40. package/storages/batched_adds.js +73 -0
  41. package/storages/dataset.d.ts +13 -8
  42. package/storages/dataset.js +149 -40
  43. package/storages/index.d.ts +4 -4
  44. package/storages/index.js +2 -4
  45. package/storages/key_value_store.d.ts +16 -35
  46. package/storages/key_value_store.js +223 -110
  47. package/storages/key_value_store_codec.js +6 -11
  48. package/storages/request_dedup_cache.d.ts +1 -4
  49. package/storages/request_dedup_cache.js +15 -15
  50. package/storages/request_list.d.ts +9 -104
  51. package/storages/request_list.js +236 -233
  52. package/storages/request_loader.d.ts +49 -18
  53. package/storages/request_loader.js +36 -1
  54. package/storages/request_manager.d.ts +86 -0
  55. package/storages/request_manager_tandem.d.ts +14 -38
  56. package/storages/request_manager_tandem.js +67 -64
  57. package/storages/request_queue.d.ts +23 -50
  58. package/storages/request_queue.js +371 -226
  59. package/storages/storage_instance_manager.d.ts +2 -4
  60. package/storages/storage_instance_manager.js +21 -21
  61. package/storages/storage_stats.d.ts +1 -1
  62. package/storages/storage_stats.js +4 -4
  63. package/storages/transaction.d.ts +270 -0
  64. package/storages/transaction.js +296 -0
  65. package/storages/utils.d.ts +6 -3
  66. package/storages/utils.js +11 -2
  67. package/system-info/runtime.js +7 -7
  68. package/url.d.ts +9 -0
  69. package/url.js +11 -0
  70. package/validators.d.ts +23 -25
  71. package/validators.js +14 -25
  72. package/autoscaling/autoscaled_pool.d.ts +0 -213
  73. package/autoscaling/autoscaled_pool.js +0 -378
  74. package/autoscaling/client_load_signal.d.ts +0 -59
  75. package/autoscaling/client_load_signal.js +0 -73
  76. package/autoscaling/concurrency_system.d.ts +0 -283
  77. package/autoscaling/concurrency_system.js +0 -350
  78. package/autoscaling/cpu_load_signal.d.ts +0 -44
  79. package/autoscaling/cpu_load_signal.js +0 -46
  80. package/autoscaling/event_loop_load_signal.d.ts +0 -54
  81. package/autoscaling/event_loop_load_signal.js +0 -60
  82. package/autoscaling/index.d.ts +0 -9
  83. package/autoscaling/index.js +0 -9
  84. package/autoscaling/load_signal.d.ts +0 -99
  85. package/autoscaling/load_signal.js +0 -103
  86. package/autoscaling/memory_load_signal.d.ts +0 -56
  87. package/autoscaling/memory_load_signal.js +0 -106
  88. package/autoscaling/snapshotter.d.ts +0 -87
  89. package/autoscaling/snapshotter.js +0 -67
  90. package/autoscaling/system_status.d.ts +0 -161
  91. package/autoscaling/system_status.js +0 -139
  92. package/autoscaling/weighted_avg.d.ts +0 -5
  93. package/autoscaling/weighted_avg.js +0 -14
  94. package/cookie_utils.d.ts +0 -44
  95. package/cookie_utils.js +0 -122
  96. package/crawlers/context_pipeline.d.ts +0 -70
  97. package/crawlers/context_pipeline.js +0 -122
  98. package/crawlers/crawler_commons.d.ts +0 -257
  99. package/crawlers/crawler_commons.js +0 -107
  100. package/crawlers/error_snapshotter.d.ts +0 -59
  101. package/crawlers/error_snapshotter.js +0 -117
  102. package/crawlers/error_tracker.d.ts +0 -54
  103. package/crawlers/error_tracker.js +0 -308
  104. package/crawlers/index.d.ts +0 -5
  105. package/crawlers/index.js +0 -5
  106. package/crawlers/internals/types.d.ts +0 -7
  107. package/crawlers/statistics.d.ts +0 -209
  108. package/crawlers/statistics.js +0 -350
  109. package/enqueue_links/enqueue_links.d.ts +0 -264
  110. package/enqueue_links/enqueue_links.js +0 -271
  111. package/enqueue_links/index.d.ts +0 -2
  112. package/enqueue_links/index.js +0 -2
  113. package/enqueue_links/shared.d.ts +0 -83
  114. package/enqueue_links/shared.js +0 -221
  115. package/router.d.ts +0 -309
  116. package/router.js +0 -309
  117. package/session_pool/consts.d.ts +0 -3
  118. package/session_pool/consts.js +0 -3
  119. package/session_pool/errors.d.ts +0 -7
  120. package/session_pool/errors.js +0 -11
  121. package/session_pool/fingerprint.d.ts +0 -9
  122. package/session_pool/fingerprint.js +0 -30
  123. package/session_pool/index.d.ts +0 -4
  124. package/session_pool/index.js +0 -4
  125. package/session_pool/session.d.ts +0 -161
  126. package/session_pool/session.js +0 -218
  127. package/session_pool/session_pool.d.ts +0 -246
  128. package/session_pool/session_pool.js +0 -386
  129. package/storages/access_checking.d.ts +0 -12
  130. package/storages/access_checking.js +0 -17
  131. package/storages/sitemap_request_loader.d.ts +0 -249
  132. package/storages/sitemap_request_loader.js +0 -432
  133. /package/{crawlers/internals/types.js → events/system_info.js} +0 -0
@@ -1,13 +1,17 @@
1
1
  import { inspect } from 'node:util';
2
- import { downloadListOfUrls, isAsyncIterable, isIterable, sleep } from '@crawlee/utils';
3
- import ow from 'ow';
2
+ import { isAsyncIterable, isIterable } from '@crawlee/utils/internal';
3
+ import { downloadListOfUrls } from '@crawlee/utils';
4
+ import { z } from 'zod';
4
5
  import { LruCache } from '@apify/datastructures';
6
+ import { tryCancel } from '@apify/timeout';
5
7
  import { Configuration } from '../configuration.js';
6
8
  import { getObjectType } from '../debug.js';
7
- import { chunkedAsyncIterable, peekableAsyncIterable } from '../iterables.js';
9
+ import { EventType } from '../events/event_manager.js';
8
10
  import { Request } from '../request.js';
9
11
  import { serviceLocator } from '../service_locator.js';
10
- import { checkStorageAccess } from './access_checking.js';
12
+ import { parseArgument, schemas, validators } from '../validators.js';
13
+ import { activeStorageTransaction, rejectOperationInTransaction } from './transaction.js';
14
+ import { drainRequestBatches } from './batched_adds.js';
11
15
  import { StorageStatsTracker } from './storage_stats.js';
12
16
  import { resolveStorageIdentifier } from './storage_instance_manager.js';
13
17
  import { getRequestId, purgeDefaultStorages } from './utils.js';
@@ -17,6 +21,44 @@ import { RequestDeduplicationCache } from './request_dedup_cache.js';
17
21
  * @internal
18
22
  */
19
23
  const MAX_CACHED_REQUESTS = 2_000_000;
24
+ const iterableSchema = z.custom((value) => isIterable(value) || isAsyncIterable(value), {
25
+ error: (issue) => `Expected an iterable or async iterable, got ${getObjectType(issue.input)}`,
26
+ });
27
+ const operationOptionsSchema = z.strictObject({
28
+ forefront: z.boolean().default(false),
29
+ });
30
+ const addRequestsOptionsSchema = z.strictObject({
31
+ forefront: z.boolean().default(false),
32
+ cache: z.boolean().default(true),
33
+ });
34
+ const addRequestsBatchedOptionsSchema = z.strictObject({
35
+ forefront: z.boolean().optional(),
36
+ waitForAllRequestsToBeAdded: z.boolean().default(false),
37
+ batchSize: schemas.anyNumber.default(1000),
38
+ waitBetweenBatchesMillis: schemas.anyNumber.default(1000),
39
+ maxNewRequests: schemas.anyNumber.optional(),
40
+ });
41
+ // Compiled: these run once per request.
42
+ const newRequestLikeSchema = z.compile(z.looseObject({
43
+ url: z.string(),
44
+ id: z.undefined().optional(),
45
+ }));
46
+ const handledRequestSchema = z.compile(z.looseObject({
47
+ id: z.string(),
48
+ uniqueKey: z.string(),
49
+ handledAt: z.string().optional(),
50
+ }));
51
+ const reclaimedRequestSchema = z.compile(z.looseObject({
52
+ id: z.string(),
53
+ uniqueKey: z.string(),
54
+ }));
55
+ const uniqueKeySchema = z.string();
56
+ const openOptionsSchema = z.strictObject({
57
+ configuration: z.instanceof(Configuration).optional(),
58
+ storageBackend: validators.storageBackend.optional(),
59
+ proxyConfiguration: validators.proxyConfiguration.optional(),
60
+ httpClient: schemas.httpClient.optional(),
61
+ });
20
62
  /**
21
63
  * Represents a queue of URLs to crawl, which is used for deep crawling of websites
22
64
  * where you start with several URLs and then recursively
@@ -55,26 +97,26 @@ export class RequestQueue {
55
97
  id;
56
98
  name;
57
99
  backend;
58
- proxyConfiguration;
100
+ #proxyConfiguration;
59
101
  log;
60
- requestCache;
102
+ #requestCache;
61
103
  /**
62
104
  * Remembers the `requestId` of every request already submitted to the client — including background
63
105
  * batches that `requestCache` skips — so overlapping URL sets aren't re-submitted.
64
106
  * See {@link RequestDeduplicationCache} for why this is a separate, cheaper cache.
65
107
  */
66
- requestSeenCache;
67
- queuePausedForMigration = false;
68
- inProgressRequestBatchCount = 0;
108
+ #requestSeenCache;
109
+ #queuePausedForMigration = false;
110
+ #inProgressRequestBatchCount = 0;
69
111
  /**
70
112
  * The largest expected request-processing time (in seconds) seen so far via
71
113
  * {@link setExpectedRequestProcessingTimeSecs}. Used to ensure that value is only ever raised, never
72
114
  * lowered, before being forwarded to the storage backend.
73
115
  */
74
- expectedRequestProcessingSecs = 0;
75
- httpClient;
76
- events;
77
- statsTracker = new StorageStatsTracker({
116
+ #expectedRequestProcessingSecs = 0;
117
+ #httpClient;
118
+ #events;
119
+ #statsTracker = new StorageStatsTracker({
78
120
  writeCount: 0,
79
121
  headItemReadCount: 0,
80
122
  });
@@ -83,7 +125,7 @@ export class RequestQueue {
83
125
  * queue-head reads issued to the underlying storage backend). Counted per backend call.
84
126
  */
85
127
  get stats() {
86
- return this.statsTracker.current;
128
+ return this.#statsTracker.current;
87
129
  }
88
130
  /**
89
131
  * @internal
@@ -91,14 +133,14 @@ export class RequestQueue {
91
133
  constructor(options) {
92
134
  this.id = options.metadata.id;
93
135
  this.name = options.metadata.name;
94
- this.events = serviceLocator.getEventManager();
136
+ this.#events = serviceLocator.getEventManager();
95
137
  this.backend = options.backend;
96
- this.proxyConfiguration = options.proxyConfiguration;
97
- this.requestCache = new LruCache({ maxLength: MAX_CACHED_REQUESTS });
98
- this.requestSeenCache = new RequestDeduplicationCache();
138
+ this.#proxyConfiguration = options.proxyConfiguration;
139
+ this.#requestCache = new LruCache({ maxLength: MAX_CACHED_REQUESTS });
140
+ this.#requestSeenCache = new RequestDeduplicationCache();
99
141
  this.log = serviceLocator.getLogger().child({ prefix: `RequestQueue(${this.id}, ${this.name ?? 'no-name'})` });
100
- this.events.on("migrating" /* EventType.MIGRATING */, async () => {
101
- this.queuePausedForMigration = true;
142
+ this.#events.on(EventType.MIGRATING, async () => {
143
+ this.#queuePausedForMigration = true;
102
144
  });
103
145
  }
104
146
  /**
@@ -134,26 +176,24 @@ export class RequestQueue {
134
176
  * @param [options] Request queue operation options.
135
177
  */
136
178
  async addRequest(requestLike, options = {}) {
137
- checkStorageAccess();
138
- ow(requestLike, ow.object);
139
- ow(options, ow.object.exactShape({
140
- forefront: ow.optional.boolean,
141
- }));
142
- const { forefront = false } = options;
179
+ const transaction = activeStorageTransaction();
180
+ parseArgument(requestLike, schemas.anyObject);
181
+ const { forefront } = parseArgument(options, operationOptionsSchema);
143
182
  if ('requestsFromUrl' in requestLike) {
144
- const requests = await this.fetchRequestsFromUrl(requestLike);
145
- const processedRequests = await this.addFetchedRequests(requestLike, requests, options);
183
+ const requests = await this.#fetchRequestsFromUrl(requestLike);
184
+ const processedRequests = await this.#addFetchedRequests(requestLike, requests, options);
146
185
  return { ...processedRequests[0], forefront };
147
186
  }
148
- ow(requestLike, ow.object.partialShape({
149
- url: ow.string,
150
- id: ow.undefined,
151
- }));
187
+ parseArgument(requestLike, newRequestLikeSchema);
152
188
  const request = requestLike instanceof Request ? requestLike : new Request(requestLike);
189
+ if (transaction?.policy.requestQueue === 'deferred') {
190
+ return this.#addRequestDeferred(transaction, request, forefront);
191
+ }
153
192
  const cacheKey = getRequestId(request.uniqueKey);
154
- const cachedInfo = this.requestCache.get(cacheKey);
193
+ const cachedInfo = this.#requestCache.get(cacheKey);
155
194
  if (cachedInfo) {
156
195
  request.id = cachedInfo.id;
196
+ this.#recordRequestJournalEntry(transaction, [request], forefront, true);
157
197
  return {
158
198
  wasAlreadyPresent: true,
159
199
  // We may assume that if request is in local cache then also the information if the
@@ -164,17 +204,163 @@ export class RequestQueue {
164
204
  forefront,
165
205
  };
166
206
  }
167
- this.statsTracker.add('writeCount');
207
+ this.#statsTracker.add('writeCount');
168
208
  const { processedRequests } = await this.backend.addBatchOfRequests([request], { forefront });
209
+ this.#recordRequestJournalEntry(transaction, [request], forefront, true);
169
210
  const queueOperationInfo = {
170
211
  ...processedRequests[0],
171
212
  uniqueKey: request.uniqueKey,
172
213
  forefront,
173
214
  };
174
- this.cacheRequest(cacheKey, queueOperationInfo);
175
- this.requestSeenCache.add(cacheKey, request.id);
215
+ this.#cacheRequest(cacheKey, queueOperationInfo);
216
+ this.#requestSeenCache.add(cacheKey, request.id);
176
217
  return queueOperationInfo;
177
218
  }
219
+ /**
220
+ * Journals an addition for introspection only; these entries are never replayed. A no-op unless the
221
+ * transaction is open, so detached and outliving writers stay out of the journal.
222
+ */
223
+ #recordRequestJournalEntry(transaction, requests, forefront, writeThrough) {
224
+ if (!transaction?.isActive || requests.length === 0)
225
+ return;
226
+ transaction.recordJournalEntry({
227
+ type: 'requestQueue',
228
+ participant: this,
229
+ requests: requests.map((request) => ({
230
+ url: request.url,
231
+ uniqueKey: request.uniqueKey,
232
+ label: request.label,
233
+ })),
234
+ forefront,
235
+ writeThrough,
236
+ });
237
+ }
238
+ /**
239
+ * The requests buffered by the given transaction for this queue, keyed by `uniqueKey` — a dedup
240
+ * index derived from the transaction journal.
241
+ */
242
+ #bufferedRequests(transaction) {
243
+ const buffered = new Map();
244
+ // Only `deferred` records snapshots, so scanning the journal under `writeThrough` never finds any.
245
+ if (transaction.policy.requestQueue !== 'deferred')
246
+ return buffered;
247
+ for (const entry of transaction.journal) {
248
+ if (entry.type !== 'requestQueue' || entry.participant !== this)
249
+ continue;
250
+ for (const request of entry.requests) {
251
+ if (request.snapshot !== undefined)
252
+ buffered.set(request.uniqueKey, request.snapshot);
253
+ }
254
+ }
255
+ return buffered;
256
+ }
257
+ /**
258
+ * Adds a request under the `deferred` policy: journaled now, really added by the commit replay.
259
+ * A new request's `requestId` is the local `uniqueKey` hash and is **provisional** — never write it
260
+ * to `request.id` or the dedup caches. Dedup is cheapest-first: buffer, caches, then a backend probe.
261
+ */
262
+ async #addRequestDeferred(transaction, request, forefront, buffered = this.#bufferedRequests(transaction)) {
263
+ // This transaction's own buffered adds; the shared caches never see them (provisional ids).
264
+ if (buffered.has(request.uniqueKey)) {
265
+ this.#recordRequestJournalEntry(transaction, [request], forefront, false);
266
+ return {
267
+ wasAlreadyPresent: true,
268
+ wasAlreadyHandled: false,
269
+ requestId: getRequestId(request.uniqueKey),
270
+ uniqueKey: request.uniqueKey,
271
+ forefront,
272
+ };
273
+ }
274
+ // The caches hold real backend ids. Only *writing* provisional ids to them would be wrong;
275
+ // reading saves a probe. Same lookup as the write-through path.
276
+ const cacheKey = getRequestId(request.uniqueKey);
277
+ const cachedInfo = this.#requestCache.get(cacheKey);
278
+ const knownRequestId = cachedInfo?.id ?? this.#requestSeenCache.get(cacheKey);
279
+ if (knownRequestId) {
280
+ this.#recordRequestJournalEntry(transaction, [request], forefront, false);
281
+ return {
282
+ wasAlreadyPresent: true,
283
+ // The dedup cache doesn't track the handled state; only the full record does.
284
+ wasAlreadyHandled: cachedInfo?.isHandled ?? false,
285
+ requestId: knownRequestId,
286
+ uniqueKey: request.uniqueKey,
287
+ forefront,
288
+ };
289
+ }
290
+ // The caches are bounded, so a miss is not proof of absence - probe for an accurate answer.
291
+ const existing = await this.backend.getRequest(request.uniqueKey);
292
+ if (existing) {
293
+ this.#recordRequestJournalEntry(transaction, [request], forefront, false);
294
+ return {
295
+ wasAlreadyPresent: true,
296
+ wasAlreadyHandled: existing.handledAt != null,
297
+ requestId: existing.id,
298
+ uniqueKey: request.uniqueKey,
299
+ forefront,
300
+ };
301
+ }
302
+ // The entry below *is* the write, so a transaction closed during the probe must not receive it -
303
+ // pass through instead, per the closed-transaction rule. Under `deferred` that can land an
304
+ // addition a rollback would have discarded; dedup bounds that cost, silent loss is unbounded.
305
+ if (!transaction.isActive) {
306
+ return await this.addRequest(request, { forefront });
307
+ }
308
+ const snapshot = JSON.parse(JSON.stringify(request));
309
+ // Strip-list, not allow-list: every user-facing field flows through, including ones added to
310
+ // `Request` in the future. The exceptions are `id` and `handledAt`, the two backend-owned
311
+ // lifecycle fields.
312
+ delete snapshot.id;
313
+ delete snapshot.handledAt;
314
+ transaction.recordJournalEntry({
315
+ type: 'requestQueue',
316
+ participant: this,
317
+ requests: [{ url: request.url, uniqueKey: request.uniqueKey, label: request.label, snapshot }],
318
+ forefront,
319
+ writeThrough: false,
320
+ });
321
+ buffered.set(request.uniqueKey, snapshot);
322
+ return {
323
+ wasAlreadyPresent: false,
324
+ wasAlreadyHandled: false,
325
+ requestId: getRequestId(request.uniqueKey),
326
+ uniqueKey: request.uniqueKey,
327
+ forefront,
328
+ };
329
+ }
330
+ /** @internal */
331
+ async commitJournalEntries(entries) {
332
+ // Replay through `backend.addBatchOfRequests`, *not* the batched frontend wrapper - the wrapper
333
+ // resolves after the first chunk and sleeps between the rest, neither of which commit may
334
+ // inherit. One call per `forefront` flag; the order of forefront additions is arbitrary anyway.
335
+ for (const forefront of [false, true]) {
336
+ const requests = entries.flatMap((entry) => entry.type === 'requestQueue' && entry.forefront === forefront
337
+ ? // Requests without a snapshot were deduplicated or written through; nothing to replay.
338
+ entry.requests
339
+ .filter((journaled) => journaled.snapshot !== undefined)
340
+ .map((journaled) => Request.fromSchema(journaled.snapshot))
341
+ : []);
342
+ if (requests.length === 0)
343
+ continue;
344
+ this.#statsTracker.add('writeCount');
345
+ const { processedRequests, unprocessedRequests } = await this.backend.addBatchOfRequests(requests, {
346
+ forefront,
347
+ });
348
+ // Only now, with the real backend-assigned ids, may the shared dedup caches be populated.
349
+ for (const processed of processedRequests) {
350
+ const cacheKey = getRequestId(processed.uniqueKey);
351
+ this.#cacheRequest(cacheKey, { ...processed, forefront });
352
+ this.#requestSeenCache.add(cacheKey, processed.requestId);
353
+ }
354
+ if (unprocessedRequests.length > 0) {
355
+ // Warn and skip, rather than retry or fail. `unprocessedRequests` is what remains after
356
+ // the backend's own transient-error handling - a semantic rejection that retrying here
357
+ // would only re-poke. And failing the commit would let one malformed request hold the
358
+ // whole transaction hostage.
359
+ this.log.warning('Some requests were rejected by the request queue while committing a storage transaction and will be skipped. ' +
360
+ "This usually means the request data is malformed (e.g. an invalid 'userData' shape).", { unprocessedRequests });
361
+ }
362
+ }
363
+ }
178
364
  /**
179
365
  * Adds requests to the queue in batches of 25. This method will wait till all the requests are added
180
366
  * to the queue before resolving. You should prefer using `queue.addRequestsBatched()` or `crawler.addRequests()`
@@ -190,15 +376,9 @@ export class RequestQueue {
190
376
  * @param [options] Request queue operation options.
191
377
  */
192
378
  async addRequests(requestsLike, options = {}) {
193
- checkStorageAccess();
194
- ow(requestsLike, ow.object
195
- .is((value) => isIterable(value) || isAsyncIterable(value))
196
- .message((value) => `Expected an iterable or async iterable, got ${getObjectType(value)}`));
197
- ow(options, ow.object.exactShape({
198
- forefront: ow.optional.boolean,
199
- cache: ow.optional.boolean,
200
- }));
201
- const { forefront = false, cache = true } = options;
379
+ const transaction = activeStorageTransaction();
380
+ parseArgument(requestsLike, iterableSchema);
381
+ const { forefront, cache } = parseArgument(options, addRequestsOptionsSchema);
202
382
  const uniqueKeyToCacheKey = new Map();
203
383
  const getCachedRequestId = (uniqueKey) => {
204
384
  const cached = uniqueKeyToCacheKey.get(uniqueKey);
@@ -218,19 +398,27 @@ export class RequestQueue {
218
398
  requests.push(new Request({ url: requestLike }));
219
399
  }
220
400
  else if ('requestsFromUrl' in requestLike) {
221
- const fetchedRequests = await this.fetchRequestsFromUrl(requestLike);
222
- await this.addFetchedRequests(requestLike, fetchedRequests, options);
401
+ const fetchedRequests = await this.#fetchRequestsFromUrl(requestLike);
402
+ await this.#addFetchedRequests(requestLike, fetchedRequests, options);
223
403
  }
224
404
  else {
225
405
  requests.push(requestLike instanceof Request ? requestLike : new Request(requestLike));
226
406
  }
227
407
  }
408
+ if (transaction?.policy.requestQueue === 'deferred') {
409
+ const buffered = this.#bufferedRequests(transaction);
410
+ for (const request of requests) {
411
+ results.processedRequests.push(await this.#addRequestDeferred(transaction, request, forefront, buffered));
412
+ }
413
+ return results;
414
+ }
415
+ this.#recordRequestJournalEntry(transaction, requests, forefront, true);
228
416
  const requestsToAdd = new Map();
229
417
  for (const request of requests) {
230
418
  const cacheKey = getCachedRequestId(request.uniqueKey);
231
419
  // Prefer the full `requestCache` record; fall back to the dedup cache for background batches it skips.
232
- const cachedInfo = this.requestCache.get(cacheKey);
233
- const knownRequestId = cachedInfo?.id ?? this.requestSeenCache.get(cacheKey);
420
+ const cachedInfo = this.#requestCache.get(cacheKey);
421
+ const knownRequestId = cachedInfo?.id ?? this.#requestSeenCache.get(cacheKey);
234
422
  if (knownRequestId) {
235
423
  request.id = knownRequestId;
236
424
  results.processedRequests.push({
@@ -249,7 +437,7 @@ export class RequestQueue {
249
437
  if (!requestsToAdd.size) {
250
438
  return results;
251
439
  }
252
- this.statsTracker.add('writeCount');
440
+ this.#statsTracker.add('writeCount');
253
441
  const apiResults = await this.backend.addBatchOfRequests([...requestsToAdd.values()], { forefront });
254
442
  // Report unprocessed requests
255
443
  results.unprocessedRequests = apiResults.unprocessedRequests;
@@ -259,10 +447,10 @@ export class RequestQueue {
259
447
  results.processedRequests.push(newRequest);
260
448
  const cacheKey = getCachedRequestId(newRequest.uniqueKey);
261
449
  if (cache) {
262
- this.cacheRequest(cacheKey, { ...newRequest, forefront });
450
+ this.#cacheRequest(cacheKey, { ...newRequest, forefront });
263
451
  }
264
452
  // Unlike `requestCache`, populate this on every batch (including background ones).
265
- this.requestSeenCache.add(cacheKey, newRequest.requestId);
453
+ this.#requestSeenCache.add(cacheKey, newRequest.requestId);
266
454
  }
267
455
  return results;
268
456
  }
@@ -276,17 +464,8 @@ export class RequestQueue {
276
464
  * @param options Options for the request queue
277
465
  */
278
466
  async addRequestsBatched(requests, options = {}) {
279
- checkStorageAccess();
280
- ow(requests, ow.object
281
- .is((value) => isIterable(value) || isAsyncIterable(value))
282
- .message((value) => `Expected an iterable or async iterable, got ${getObjectType(value)}`));
283
- ow(options, ow.object.exactShape({
284
- forefront: ow.optional.boolean,
285
- waitForAllRequestsToBeAdded: ow.optional.boolean,
286
- batchSize: ow.optional.number,
287
- waitBetweenBatchesMillis: ow.optional.number,
288
- maxNewRequests: ow.optional.number,
289
- }));
467
+ parseArgument(requests, iterableSchema);
468
+ const { forefront, waitForAllRequestsToBeAdded, batchSize, waitBetweenBatchesMillis, maxNewRequests } = parseArgument(options, addRequestsBatchedOptionsSchema);
290
469
  const addRequest = this.addRequest.bind(this);
291
470
  async function* generateRequests() {
292
471
  for await (const opts of requests) {
@@ -295,7 +474,7 @@ export class RequestQueue {
295
474
  if (opts.url !== undefined && typeof opts.url !== 'string') {
296
475
  throw new Error(`Request options are not valid, the 'url' property is not a string. Input: ${inspect(opts)}`);
297
476
  }
298
- if (opts.id !== undefined) {
477
+ if ('id' in opts && opts.id !== undefined) {
299
478
  throw new Error(`Request options are not valid, the 'id' property must not be present. Input: ${inspect(opts)}`);
300
479
  }
301
480
  if (opts.requestsFromUrl !== undefined &&
@@ -305,7 +484,7 @@ export class RequestQueue {
305
484
  }
306
485
  if (opts && typeof opts === 'object' && 'requestsFromUrl' in opts) {
307
486
  // Handle URL lists right away
308
- await addRequest(opts, { forefront: options.forefront });
487
+ await addRequest(opts, { forefront });
309
488
  }
310
489
  else {
311
490
  // Yield valid requests
@@ -313,84 +492,36 @@ export class RequestQueue {
313
492
  }
314
493
  }
315
494
  }
316
- const { batchSize = 1000, waitBetweenBatchesMillis = 1000, maxNewRequests = undefined } = options;
317
- let remainingBudget = maxNewRequests ?? Infinity;
318
- const requestsOverLimit = [];
319
- // If there's a limit on the number of added requests, do not send batches bigger than the limit
320
- const effectiveChunkSize = maxNewRequests !== undefined ? () => Math.min(batchSize, remainingBudget) : batchSize;
321
- // Hold onto the underlying iterator so we can drain leftovers from it in buildResult
322
- const requestIterator = generateRequests();
323
- const chunks = peekableAsyncIterable(chunkedAsyncIterable(requestIterator, effectiveChunkSize));
324
- const chunksIterator = chunks[Symbol.asyncIterator]();
325
- /**
326
- * Process a chunk: send it to the queue, then update the remaining budget if maxNewRequests is active.
327
- *
328
- * Requests the backend reports as unprocessed are warned about and skipped rather than retried:
329
- * `unprocessedRequests` is what remains after the backend's own transient-error handling - a
330
- * semantic rejection (e.g. a malformed `userData` shape) that re-sending would only re-poke.
331
- * Retrying transient failures is the storage backend's job, not the frontend's.
332
- */
333
- const processChunk = async (chunk, cache = true) => {
334
- const { processedRequests, unprocessedRequests } = await this.addRequests(chunk, {
335
- forefront: options.forefront,
336
- cache,
337
- });
338
- if (unprocessedRequests.length > 0) {
339
- this.log.warning('Some requests were rejected by the request queue and will be skipped. ' +
340
- "This usually means the request data is malformed (e.g. an invalid 'userData' shape).", { unprocessedRequests });
341
- }
342
- if (maxNewRequests !== undefined) {
343
- remainingBudget -= processedRequests.filter((r) => !r.wasAlreadyPresent).length;
344
- }
345
- return processedRequests;
346
- };
347
- /**
348
- * Build the final result. When maxNewRequests is set, drains any remaining items
349
- * from the underlying request iterator into requestsOverLimit.
350
- *
351
- * We accept the iterator explicitly (rather than closing over it) to make it obvious
352
- * that this is the *same* iterator that `chunkedAsyncIterable` has been consuming —
353
- * so only unconsumed items are drained. We drain `requestIterator` (not `chunks`)
354
- * because `chunkedAsyncIterable` stops yielding when the budget-based chunk size
355
- * drops to 0, leaving unconsumed items in the underlying iterator.
356
- */
357
- const buildResult = async (addedRequests, waitForAllRequestsToBeAdded, unconsumedIterator) => {
358
- if (maxNewRequests !== undefined) {
359
- for await (const request of unconsumedIterator) {
360
- requestsOverLimit.push(request);
495
+ return drainRequestBatches({
496
+ items: generateRequests(),
497
+ batchSize,
498
+ waitBetweenBatchesMillis,
499
+ waitForAllRequestsToBeAdded,
500
+ maxNewRequests,
501
+ /**
502
+ * Requests the backend reports as unprocessed are warned about and skipped rather than retried:
503
+ * `unprocessedRequests` is what remains after the backend's own transient-error handling - a
504
+ * semantic rejection (e.g. a malformed `userData` shape) that re-sending would only re-poke.
505
+ * Retrying transient failures is the storage backend's job, not the frontend's.
506
+ */
507
+ processChunk: async (chunk, isInitial) => {
508
+ const { processedRequests, unprocessedRequests } = await this.addRequests(chunk, {
509
+ forefront,
510
+ cache: isInitial,
511
+ });
512
+ if (unprocessedRequests.length > 0) {
513
+ this.log.warning('Some requests were rejected by the request queue and will be skipped. ' +
514
+ "This usually means the request data is malformed (e.g. an invalid 'userData' shape).", { unprocessedRequests });
361
515
  }
362
- }
363
- return { addedRequests, waitForAllRequestsToBeAdded, requestsOverLimit };
364
- };
365
- // Add initial batch to process right away
366
- const initialChunk = await chunksIterator.peek();
367
- if (initialChunk === undefined) {
368
- return buildResult([], Promise.resolve([]), requestIterator);
369
- }
370
- const addedRequests = await processChunk(initialChunk);
371
- await chunksIterator.next();
372
- // If we have no more requests to add (either exhausted or budget hit), return immediately
373
- if ((await chunksIterator.peek()) === undefined) {
374
- return buildResult(addedRequests, Promise.resolve([]), requestIterator);
375
- }
376
- // eslint-disable-next-line no-async-promise-executor
377
- const promise = new Promise(async (resolve) => {
378
- const finalAddedRequests = [];
379
- for await (const requestChunk of chunks) {
380
- finalAddedRequests.push(...(await processChunk(requestChunk, false)));
381
- await sleep(waitBetweenBatchesMillis);
382
- }
383
- resolve(finalAddedRequests);
384
- });
385
- this.inProgressRequestBatchCount += 1;
386
- void promise.finally(() => {
387
- this.inProgressRequestBatchCount -= 1;
516
+ return processedRequests;
517
+ },
518
+ trackBackgroundBatches: (batches) => {
519
+ this.#inProgressRequestBatchCount += 1;
520
+ void batches.finally(() => {
521
+ this.#inProgressRequestBatchCount -= 1;
522
+ });
523
+ },
388
524
  });
389
- // When maxNewRequests is set, we must wait for all batches so we can accurately report skipped requests.
390
- if (options.waitForAllRequestsToBeAdded || maxNewRequests !== undefined) {
391
- addedRequests.push(...(await promise));
392
- }
393
- return buildResult(addedRequests, promise, requestIterator);
394
525
  }
395
526
  /**
396
527
  * Gets the request from the queue specified by its `uniqueKey`.
@@ -399,12 +530,17 @@ export class RequestQueue {
399
530
  * @returns Returns the request object, or `null` if it was not found.
400
531
  */
401
532
  async getRequest(uniqueKey) {
402
- checkStorageAccess();
403
- ow(uniqueKey, ow.string);
404
- const requestOptions = await this.backend.getRequest(uniqueKey);
405
- if (!requestOptions)
533
+ const transaction = activeStorageTransaction();
534
+ parseArgument(uniqueKey, uniqueKeySchema);
535
+ // Requests buffered by the active transaction (under the `deferred` write policy) are visible to it.
536
+ const buffered = transaction && this.#bufferedRequests(transaction).get(uniqueKey);
537
+ if (buffered) {
538
+ return Request.fromSchema(buffered);
539
+ }
540
+ const schema = await this.backend.getRequest(uniqueKey);
541
+ if (!schema)
406
542
  return null;
407
- return new Request(requestOptions);
543
+ return Request.fromSchema(schema);
408
544
  }
409
545
  /**
410
546
  * Returns a next request in the queue to be processed, or `null` if there are no more pending requests.
@@ -418,21 +554,21 @@ export class RequestQueue {
418
554
  * Note that the `null` return value doesn't mean the queue processing finished,
419
555
  * it means there are currently no pending requests.
420
556
  * To check whether all requests in queue were finished,
421
- * use {@link RequestQueue.isFinished} instead.
557
+ * use {@link RequestQueue.checkReadiness} instead.
422
558
  *
423
559
  * @returns
424
560
  * Returns the request object or `null` if there are no more pending requests.
425
561
  */
426
562
  async fetchNextRequest() {
427
- checkStorageAccess();
428
- if (this.queuePausedForMigration) {
563
+ rejectOperationInTransaction('RequestQueue.fetchNextRequest()', 'it is part of the crawler request-processing bookkeeping, which a transaction must not affect.');
564
+ if (this.#queuePausedForMigration) {
429
565
  return null;
430
566
  }
431
- this.statsTracker.add('headItemReadCount');
432
- const requestOptions = await this.backend.fetchNextRequest();
433
- if (!requestOptions)
567
+ this.#statsTracker.add('headItemReadCount');
568
+ const schema = await this.backend.fetchNextRequest();
569
+ if (!schema)
434
570
  return null;
435
- return new Request(requestOptions);
571
+ return Request.fromSchema(schema);
436
572
  }
437
573
  /**
438
574
  * Marks a request that was previously returned by the
@@ -441,15 +577,11 @@ export class RequestQueue {
441
577
  * Handled requests will never again be returned by the `fetchNextRequest` function.
442
578
  */
443
579
  async markRequestAsHandled(request) {
444
- checkStorageAccess();
445
- ow(request, ow.object.partialShape({
446
- id: ow.string,
447
- uniqueKey: ow.string,
448
- handledAt: ow.optional.string,
449
- }));
450
- const forefront = this.requestCache.get(getRequestId(request.uniqueKey))?.forefront ?? false;
580
+ rejectOperationInTransaction('RequestQueue.markRequestAsHandled()', 'it is part of the crawler request-processing bookkeeping, which a transaction must not affect.');
581
+ parseArgument(request, handledRequestSchema);
582
+ const forefront = this.#requestCache.get(getRequestId(request.uniqueKey))?.forefront ?? false;
451
583
  const handledAt = request.handledAt ?? new Date().toISOString();
452
- this.statsTracker.add('writeCount');
584
+ this.#statsTracker.add('writeCount');
453
585
  const processedRequest = await this.backend.markRequestAsHandled({
454
586
  ...request,
455
587
  handledAt,
@@ -464,7 +596,7 @@ export class RequestQueue {
464
596
  uniqueKey: request.uniqueKey,
465
597
  forefront,
466
598
  };
467
- this.cacheRequest(getRequestId(request.uniqueKey), queueOperationInfo);
599
+ this.#cacheRequest(getRequestId(request.uniqueKey), queueOperationInfo);
468
600
  return queueOperationInfo;
469
601
  }
470
602
  /**
@@ -474,17 +606,13 @@ export class RequestQueue {
474
606
  * For example, this lets you store the number of retries or error messages for the request.
475
607
  */
476
608
  async reclaimRequest(request, options = {}) {
477
- checkStorageAccess();
478
- ow(request, ow.object.partialShape({
479
- id: ow.string,
480
- uniqueKey: ow.string,
481
- }));
482
- ow(options, ow.object.exactShape({
483
- forefront: ow.optional.boolean,
484
- }));
485
- const { forefront = false } = options;
486
- this.statsTracker.add('writeCount');
487
- const processedRequest = await this.backend.reclaimRequest(request, { forefront });
609
+ rejectOperationInTransaction('RequestQueue.reclaimRequest()', 'it is part of the crawler request-processing bookkeeping, which a transaction must not affect.');
610
+ parseArgument(request, reclaimedRequestSchema);
611
+ const { forefront } = parseArgument(options, operationOptionsSchema);
612
+ this.#statsTracker.add('writeCount');
613
+ const processedRequest = await this.backend.reclaimRequest(request, {
614
+ forefront,
615
+ });
488
616
  // The request was not in progress — nothing to reclaim.
489
617
  if (!processedRequest) {
490
618
  return null;
@@ -494,38 +622,41 @@ export class RequestQueue {
494
622
  uniqueKey: request.uniqueKey,
495
623
  forefront,
496
624
  };
497
- this.cacheRequest(getRequestId(request.uniqueKey), queueOperationInfo);
625
+ this.#cacheRequest(getRequestId(request.uniqueKey), queueOperationInfo);
498
626
  return queueOperationInfo;
499
627
  }
500
628
  /**
501
- * Resolves to `true` if the next call to {@link RequestQueue.fetchNextRequest} would return
502
- * `null`, i.e. there are no pending requests to fetch right now. Otherwise it resolves to `false`.
503
- *
504
- * Note that even if the queue is empty, there might be some requests currently being processed
505
- * (fetched but not yet handled or reclaimed). An empty queue therefore does not mean crawling is
506
- * finished — those in-progress requests may still be reclaimed, and background tasks may still be
507
- * adding more requests. To check whether all activity in the queue has finished, use
508
- * {@link RequestQueue.isFinished}.
629
+ * A queue hands requests out as fast as they are asked for; pacing is a job for a manager wrapped around it,
630
+ * such as {@link ThrottlingRequestManager}.
631
+ * @inheritdoc
509
632
  */
510
- async isEmpty() {
511
- checkStorageAccess();
512
- return this.backend.isEmpty();
633
+ recordPacingSignal(_signal) {
634
+ return false;
513
635
  }
514
636
  /**
515
- * Resolves to `true` if all requests were already handled and there are no more left — including no
516
- * requests currently in progress (fetched but not yet handled or reclaimed, including requests
517
- * locked by other clients sharing the same queue) and no background add operations still in flight.
637
+ * Reports whether the queue has a request to hand over, is waiting on one, or is done.
638
+ *
639
+ * `waiting` means requests are in progress (fetched but not yet handled or reclaimed, possibly by another
640
+ * client sharing the queue) or a background add is still landing; neither has a clock, so no `readyAt`.
518
641
  *
519
- * Due to the nature of distributed storage used by the queue, the function may occasionally return
520
- * a false negative, but it shall never return a false positive.
642
+ * Due to the nature of distributed storage used by the queue, `finished` may occasionally arrive a probe or
643
+ * two late, but it is never reported early.
521
644
  */
522
- async isFinished() {
523
- checkStorageAccess();
645
+ async checkReadiness() {
646
+ const transaction = activeStorageTransaction();
647
+ // Requests buffered by the active transaction count as pending from its point of view.
648
+ if (transaction && this.#bufferedRequests(transaction).size > 0) {
649
+ return { status: 'ready' };
650
+ }
651
+ // Something fetchable outranks everything below, so this is the only backend call a probe needs.
652
+ if (!(await this.backend.isEmpty())) {
653
+ return { status: 'ready' };
654
+ }
524
655
  // We are not finished if we're still adding new requests in the background.
525
- if (this.inProgressRequestBatchCount > 0) {
526
- return false;
656
+ if (this.#inProgressRequestBatchCount > 0) {
657
+ return { status: 'waiting' };
527
658
  }
528
- return this.backend.isFinished();
659
+ return (await this.backend.isFinished()) ? { status: 'finished' } : { status: 'waiting' };
529
660
  }
530
661
  /**
531
662
  * Tells the queue how long a consumer expects to hold a fetched request before marking it handled
@@ -538,24 +669,34 @@ export class RequestQueue {
538
669
  * short the reservation of a long-lived one and have its in-flight request stolen.
539
670
  */
540
671
  async setExpectedRequestProcessingTimeSecs(secs) {
541
- if (secs <= this.expectedRequestProcessingSecs) {
672
+ if (secs <= this.#expectedRequestProcessingSecs) {
542
673
  return;
543
674
  }
544
- this.expectedRequestProcessingSecs = secs;
675
+ this.#expectedRequestProcessingSecs = secs;
545
676
  await this.backend.setExpectedRequestProcessingTimeSecs?.(secs);
546
677
  }
678
+ /**
679
+ * @inheritdoc
680
+ * Unlike {@link RequestQueue.setExpectedRequestProcessingTimeSecs}, which sizes every future lock,
681
+ * this only touches the one request it is given.
682
+ */
683
+ async extendRequestProcessingTimeSecs(request, secs) {
684
+ if (!request.id) {
685
+ return false;
686
+ }
687
+ return (await this.backend.extendRequestProcessingTimeSecs?.(request.id, secs)) ?? false;
688
+ }
547
689
  /**
548
690
  * Caches information about request to beware of unneeded addRequest() calls.
549
691
  */
550
- cacheRequest(cacheKey, queueOperationInfo) {
692
+ #cacheRequest(cacheKey, queueOperationInfo) {
551
693
  // Remove the previous entry, as otherwise our cache will never update 👀
552
- this.requestCache.remove(cacheKey);
553
- this.requestCache.add(cacheKey, {
694
+ this.#requestCache.remove(cacheKey);
695
+ this.#requestCache.add(cacheKey, {
554
696
  id: queueOperationInfo.requestId,
555
697
  isHandled: queueOperationInfo.wasAlreadyHandled,
556
698
  uniqueKey: queueOperationInfo.uniqueKey,
557
699
  hydrated: null,
558
- lockExpiresAt: null,
559
700
  forefront: queueOperationInfo.forefront,
560
701
  });
561
702
  }
@@ -564,7 +705,7 @@ export class RequestQueue {
564
705
  * depending on the mode of operation.
565
706
  */
566
707
  async drop() {
567
- checkStorageAccess();
708
+ rejectOperationInTransaction('RequestQueue.drop()');
568
709
  await this.backend.drop();
569
710
  serviceLocator.getStorageInstanceManager().removeFromCache(this);
570
711
  }
@@ -573,16 +714,16 @@ export class RequestQueue {
573
714
  * so it can be reused (e.g. across multiple `crawler.run()` calls).
574
715
  */
575
716
  async purge() {
576
- checkStorageAccess();
717
+ rejectOperationInTransaction('RequestQueue.purge()');
577
718
  await this.backend.purge();
578
719
  // Reset in-memory bookkeeping so the queue behaves as if freshly opened.
579
- this.requestCache.clear();
580
- this.requestSeenCache.clear();
581
- this.inProgressRequestBatchCount = 0;
720
+ this.#requestCache.clear();
721
+ this.#requestSeenCache.clear();
722
+ this.#inProgressRequestBatchCount = 0;
582
723
  // Reset the expected-processing-time high-water mark too, otherwise the monotonic-raise guard
583
724
  // in `setExpectedRequestProcessingTimeSecs` would let a value raised in an earlier run leak into a
584
725
  // later one and silently swallow a lower hint (the queue is meant to be reusable across runs).
585
- this.expectedRequestProcessingSecs = 0;
726
+ this.#expectedRequestProcessingSecs = 0;
586
727
  }
587
728
  /**
588
729
  * @inheritdoc
@@ -630,21 +771,30 @@ export class RequestQueue {
630
771
  * @throws If the underlying storage no longer exists (e.g. it was deleted externally).
631
772
  */
632
773
  async getInfo() {
633
- checkStorageAccess();
634
- return this.backend.getMetadata();
774
+ const transaction = activeStorageTransaction();
775
+ const metadata = await this.backend.getMetadata();
776
+ const bufferedCount = transaction ? this.#bufferedRequests(transaction).size : 0;
777
+ if (bufferedCount > 0) {
778
+ return {
779
+ ...metadata,
780
+ totalRequestCount: metadata.totalRequestCount + bufferedCount,
781
+ pendingRequestCount: metadata.pendingRequestCount + bufferedCount,
782
+ };
783
+ }
784
+ return metadata;
635
785
  }
636
786
  /**
637
787
  * Fetches URLs from requestsFromUrl and returns them in format of list of requests
638
788
  */
639
- async fetchRequestsFromUrl(source) {
789
+ async #fetchRequestsFromUrl(source) {
640
790
  const { requestsFromUrl, regex, ...sharedOpts } = source;
641
791
  // Download remote resource and parse URLs.
642
792
  let urlsArr;
643
793
  try {
644
- urlsArr = await this._downloadListOfUrls({
794
+ urlsArr = await this.#downloadListOfUrls({
645
795
  url: requestsFromUrl,
646
796
  urlRegExp: regex,
647
- proxyUrl: (await this.proxyConfiguration?.newProxyInfo())?.url,
797
+ proxyUrl: (await this.#proxyConfiguration?.newProxyInfo())?.url,
648
798
  });
649
799
  }
650
800
  catch (err) {
@@ -660,7 +810,7 @@ export class RequestQueue {
660
810
  /**
661
811
  * Adds all fetched requests from a URL from a remote resource.
662
812
  */
663
- async addFetchedRequests(source, fetchedRequests, options) {
813
+ async #addFetchedRequests(source, fetchedRequests, options) {
664
814
  const { requestsFromUrl, regex } = source;
665
815
  const { addedRequests } = await this.addRequestsBatched(fetchedRequests, options);
666
816
  this.log.info('Fetched and loaded Requests from a remote resource.', {
@@ -676,10 +826,10 @@ export class RequestQueue {
676
826
  /**
677
827
  * @internal wraps public utility for mocking purposes
678
828
  */
679
- async _downloadListOfUrls(options) {
829
+ async #downloadListOfUrls(options) {
680
830
  return downloadListOfUrls({
681
831
  ...options,
682
- httpClient: this.httpClient,
832
+ httpClient: this.#httpClient,
683
833
  });
684
834
  }
685
835
  /**
@@ -700,15 +850,10 @@ export class RequestQueue {
700
850
  * @param [options] Open Request Queue options.
701
851
  */
702
852
  static async open(identifier, options = {}) {
703
- checkStorageAccess();
704
- ow(options, ow.object.exactShape({
705
- configuration: ow.optional.object.instanceOf(Configuration),
706
- storageBackend: ow.optional.object,
707
- proxyConfiguration: ow.optional.object,
708
- httpClient: ow.optional.object,
709
- }));
710
- const storageBackend = options.storageBackend ?? serviceLocator.getStorageBackend();
711
- const configuration = options.configuration ?? serviceLocator.getConfiguration();
853
+ tryCancel();
854
+ const parsedOptions = parseArgument(options, openOptionsSchema);
855
+ const storageBackend = parsedOptions.storageBackend ?? serviceLocator.getStorageBackend();
856
+ const configuration = parsedOptions.configuration ?? serviceLocator.getConfiguration();
712
857
  await purgeDefaultStorages({ onlyPurgeOnce: true, storageBackend, configuration });
713
858
  const resolved = await resolveStorageIdentifier(identifier, storageBackend, 'RequestQueue');
714
859
  const queue = await serviceLocator
@@ -718,8 +863,8 @@ export class RequestQueue {
718
863
  backendOpener: () => storageBackend.createRequestQueueBackend(resolved),
719
864
  backendCacheKey: storageBackend.getStorageBackendCacheKey?.() ?? storageBackend.constructor.name,
720
865
  });
721
- queue.proxyConfiguration = options.proxyConfiguration;
722
- queue.httpClient = options.httpClient;
866
+ queue.#proxyConfiguration = parsedOptions.proxyConfiguration;
867
+ queue.#httpClient = parsedOptions.httpClient;
723
868
  return queue;
724
869
  }
725
870
  }